[llvm] [LV] Support tail-folded epilogue loops (PR #208764)

Hassnaa Hamdi via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 28 03:25:47 PDT 2026


https://github.com/hassnaaHamdi updated https://github.com/llvm/llvm-project/pull/208764

>From 342270c7d9c9a6119d2b44aea6e1a0aaf38f91af Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Fri, 10 Jul 2026 16:06:34 +0100
Subject: [PATCH 01/25] [LV][EpilogueTailFolding] hack patch to start by
 codegen

---
 .../Vectorize/LoopVectorizationPlanner.h      |  23 +-
 .../Transforms/Vectorize/LoopVectorize.cpp    | 197 ++++++++++++++----
 llvm/lib/Transforms/Vectorize/VPlan.cpp       |  25 ++-
 llvm/lib/Transforms/Vectorize/VPlan.h         |   9 +
 .../Vectorize/VPlanConstruction.cpp           |  18 +-
 .../AArch64/fold-epilogue-tail.ll             |  63 ++++++
 .../LoopVectorize/fold-epilogue-tail.ll       |   3 +-
 7 files changed, 282 insertions(+), 56 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 3c07a6e159656..1f69de6d42e53 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -948,16 +948,17 @@ class LoopVectorizationPlanner {
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// interleaving should be avoided up-front, no plans are generated.
-  void plan(ElementCount UserVF, unsigned UserIC);
+  void plan(ElementCount UserVF, unsigned UserIC, bool IsEpilogueTFEnabled);
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
-  VPlan &getPlanFor(ElementCount VF) const;
+  VPlan &getPlanFor(ElementCount VF, bool TF) const;
 
   /// Compute and return the most profitable vectorization factor and the
   /// corresponding best VPlan. Also collect all profitable VFs in
   /// ProfitableVFs.
-  std::pair<VectorizationFactor, VPlan *> computeBestVF();
+  std::pair<VectorizationFactor, VPlan *>
+  computeBestVF(bool IsEpilogueTFEnabled);
 
   /// \return The desired interleave count.
   /// If interleave count has been specified by metadata it will be returned.
@@ -983,6 +984,7 @@ class LoopVectorizationPlanner {
   DenseMap<const SCEV *, Value *>
   executePlan(ElementCount VF, unsigned UF, VPlan &BestPlan,
               InnerLoopVectorizer &LB, DominatorTree *DT,
+              bool IsEpilogueTFEnabled,
               EpilogueVectorizationKind EpilogueVecKind =
                   EpilogueVectorizationKind::None);
 
@@ -992,10 +994,7 @@ class LoopVectorizationPlanner {
 
   /// Look through the existing plans and return true if we have one with
   /// vectorization factor \p VF.
-  bool hasPlanWithVF(ElementCount VF) const {
-    return any_of(VPlans,
-                  [&](const VPlanPtr &Plan) { return Plan->hasVF(VF); });
-  }
+  bool hasPlanWithVF(ElementCount VF, bool TF) const;
 
   /// Test a \p Predicate on a \p Range of VF's. Return the value of applying
   /// \p Predicate on Range.Start, possibly decreasing Range.End such that the
@@ -1009,10 +1008,13 @@ class LoopVectorizationPlanner {
   /// Returns nullptr if epilogue vectorization is not supported or not
   /// profitable for the loop. \p ScalarEpilogueAllowed indicates whether the
   /// epilogue lowering policy permits creating a scalar epilogue at all.
+  /// \p IsEpilogueTFEnabled indicates whether the epilogue loop should be
+  /// tail-folded rather than left with a scalar epilogue of its own.
   std::unique_ptr<VPlan> selectBestEpiloguePlan(VPlan &MainPlan,
                                                 ElementCount MainLoopVF,
                                                 unsigned IC,
-                                                bool ScalarEpilogueAllowed);
+                                                bool ScalarEpilogueAllowed,
+                                                bool IsEpilogueTFEnabled);
 
   /// Emit remarks for recipes with invalid costs in the available VPlans.
   void emitInvalidCostRemarks(OptimizationRemarkEmitter *ORE);
@@ -1046,7 +1048,7 @@ class LoopVectorizationPlanner {
   /// Build an initial VPlan, with HCFG wrapping the original scalar loop and
   /// scalar transformations applied. Returns null if an initial VPlan cannot
   /// be built.
-  VPlanPtr tryToBuildVPlan1();
+  VPlanPtr tryToBuildVPlan1(bool IsEpilogueTFEnabled);
 
   /// Build a VPlan using VPRecipes according to the information gathered by
   /// Legal and VPlan-based analysis. For outer loops, performs basic recipe
@@ -1061,7 +1063,8 @@ class LoopVectorizationPlanner {
   /// Build VPlans for power-of-2 VF's between \p MinVF and \p MaxVF inclusive,
   /// based on \p VPlan1 and according to the information gathered by Legal
   /// when it checked if it is legal to vectorize the loop.
-  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF);
+  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
+                   bool IsEpilogueTFEnabled);
 
   /// Add ComputeReductionResult recipes to the middle block to compute the
   /// final reduction results. Add Select recipes to the latch block when
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index e934fec366331..e2f57c97af55c 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3164,6 +3164,8 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
   using RecipeVFPair = std::pair<VPRecipeBase *, ElementCount>;
   SmallVector<RecipeVFPair> InvalidCosts;
   for (const auto &Plan : VPlans) {
+    if (!Plan->isCompatibleWithTF(CM->foldTailByMasking()))
+      continue;
     for (ElementCount VF : Plan->vectorFactors()) {
       // The VPlan-based cost model is designed for computing vector cost.
       // Querying VPlan-based cost model with a scarlar VF will cause some
@@ -3512,7 +3514,7 @@ bool VFSelectionContext::isEpilogueVectorizationProfitable(
 
 std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC,
-    bool ScalarEpilogueAllowed) {
+    bool ScalarEpilogueAllowed, bool IsEpilogueTFEnabled) {
   if (!EnableEpilogueVectorization) {
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization is disabled.\n");
     return nullptr;
@@ -3552,9 +3554,12 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     }
 
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization factor is forced.\n");
-    if (hasPlanWithVF(EpilogueVectorizationForceVF)) {
+    if (hasPlanWithVF(EpilogueVectorizationForceVF,
+                      MainPlan.hasTailFolded() || IsEpilogueTFEnabled)) {
       std::unique_ptr<VPlan> Clone(
-          getPlanFor(EpilogueVectorizationForceVF).duplicate());
+          getPlanFor(EpilogueVectorizationForceVF,
+                    MainPlan.hasTailFolded() || IsEpilogueTFEnabled)
+              .duplicate());
       Clone->setVF(EpilogueVectorizationForceVF);
       return Clone;
     }
@@ -3645,10 +3650,12 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
   VPlan *BestPlan = nullptr;
   for (auto &NextVF : ProfitableVFs) {
     // Skip candidate VFs without a corresponding VPlan.
-    if (!hasPlanWithVF(NextVF.Width))
+    if (!hasPlanWithVF(NextVF.Width,
+                       MainPlan.hasTailFolded() || IsEpilogueTFEnabled))
       continue;
 
-    VPlan &CurrentPlan = getPlanFor(NextVF.Width);
+    VPlan &CurrentPlan = getPlanFor(
+        NextVF.Width, MainPlan.hasTailFolded() || IsEpilogueTFEnabled);
     ElementCount EffectiveVF = GetEffectiveVF(CurrentPlan, NextVF.Width);
     // Skip fixed vector VFs > than the estimated runtime VF, or any VF > than
     // the VF of the main loop.
@@ -5342,7 +5349,8 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
   }
 }
 
-void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
+void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
+                                    bool IsEpilogueTFEnabled) {
   CM->collectValuesToIgnore();
   Config.collectElementTypesForWidening(&CM->ValuesToIgnore);
 
@@ -5357,7 +5365,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   if (MaxFactors.FixedVF.isVector() || MaxFactors.ScalableVF.isVector())
     Legal->collectUnitStridePredicates();
 
-  auto VPlan1 = tryToBuildVPlan1();
+  auto VPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ false);
   if (!VPlan1)
     return;
 
@@ -5366,7 +5374,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
     // plan for that VF only.
     ElementCount VF =
         MaxFactors.FixedVF ? MaxFactors.FixedVF : MaxFactors.ScalableVF;
-    buildVPlans(*VPlan1, VF, VF);
+    buildVPlans(*VPlan1, VF, VF, /*IsEpilogueTFEnabled*/ false);
     LLVM_DEBUG(printPlans(dbgs()));
     return;
   }
@@ -5405,12 +5413,13 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
       CM->collectNonVectorizedAndSetWideningDecisions(UserVF);
-      buildVPlans(*VPlan1, UserVF, UserVF);
+      buildVPlans(*VPlan1, UserVF, UserVF, /*IsEpilogueTFEnabled*/ false);
       ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
         CM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF);
+        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF,
+                    /*IsEpilogueTFEnabled*/ false);
       }
       if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
@@ -5442,10 +5451,27 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
     CM->collectNonVectorizedAndSetWideningDecisions(VF);
   }
 
-  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF);
-  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF);
-
+  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF,
+              /*IsEpilogueTFEnabled*/ false);
+  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
+              /*IsEpilogueTFEnabled*/ false);
   LLVM_DEBUG(printPlans(dbgs()));
+
+  // Build tail-folded vplans when IsEpilogueTFEnabled is enabled:
+  if (IsEpilogueTFEnabled) {
+    auto TFVPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ true);
+    if (!TFVPlan1)
+      return;
+    buildVPlans(*TFVPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF,
+                /*IsEpilogueTFEnabled*/ true);
+    buildVPlans(*TFVPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
+                /*IsEpilogueTFEnabled*/ true);
+    LLVM_DEBUG(dbgs() << "LV: Tail-folded vplans:\n");
+    for (auto &vplan : VPlans) {
+      if (vplan->isCompatibleWithTF(true))
+        LLVM_DEBUG(vplan->dump());
+    }
+  }
 }
 
 VPCostContext::VPCostContext(const TargetLibraryInfo &TLI, const VPlan &Plan,
@@ -5629,7 +5655,7 @@ InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
 }
 
 std::pair<VectorizationFactor, VPlan *>
-LoopVectorizationPlanner::computeBestVF() {
+LoopVectorizationPlanner::computeBestVF(bool IsEpilogueTFEnabled) {
   if (VPlans.empty())
     return {VectorizationFactor::Disabled(), nullptr};
   // If there is a single VPlan with a single VF, return it directly.
@@ -5639,13 +5665,15 @@ LoopVectorizationPlanner::computeBestVF() {
   if (VPlans.size() == 1) {
     // For outer loops, the plan has a single vector VF determined by the
     // heuristic.
-    assert((FirstPlan.hasScalarVFOnly() || hasPlanWithVF(UserVF) ||
+    assert((FirstPlan.hasScalarVFOnly() ||
+            hasPlanWithVF(UserVF, CM->foldTailByMasking()) ||
             FirstPlan.isOuterLoop()) &&
            "must have a single scalar VF, UserVF or an outer loop");
     return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
   }
 
-  if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF() && VPlans.size() == 2) {
+  if (hasPlanWithVF(UserVF, CM->foldTailByMasking()) && hasForcedEpilogueVF() &&
+      VPlans.size() == 2) {
     assert(VPlans[0]->getSingleVF() == UserVF &&
            "expected second plan to be for the forced UserVF");
     assert(VPlans[1]->getSingleVF() == EpilogueVectorizationForceVF &&
@@ -5730,9 +5758,11 @@ LoopVectorizationPlanner::computeBestVF() {
           cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
-      if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
-        BestFactor = CurrentFactor;
-        PlanForBestVF = P.get();
+      if (P->isCompatibleWithTF(CM->foldTailByMasking())) {
+        if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
+          BestFactor = CurrentFactor;
+          PlanForBestVF = P.get();
+        }
       }
 
       // If profitable add it to ProfitableVF list.
@@ -5767,7 +5797,7 @@ void LoopVectorizationPlanner::clearCostModel() { CM.reset(); }
 
 DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
     ElementCount BestVF, unsigned BestUF, VPlan &BestVPlan,
-    InnerLoopVectorizer &ILV, DominatorTree *DT,
+    InnerLoopVectorizer &ILV, DominatorTree *DT, bool IsEpilogueTFEnabled,
     EpilogueVectorizationKind EpilogueVecKind) {
   assert(BestVPlan.hasVF(BestVF) &&
          "Trying to execute plan with unsupported VF");
@@ -6412,7 +6442,7 @@ static bool verifyExecutionFrequenciesMatchBFI(VPlan &Plan, Loop *OrigLoop,
 }
 #endif
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(bool IsEpilogueTFEnabled) {
   bool IsInnerLoop = OrigLoop->isInnermost();
 
   // Set up loop versioning for inner loops with memory runtime checks.
@@ -6501,7 +6531,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
 
   RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
                  getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
-  if (CM->foldTailByMasking())
+  if (CM->foldTailByMasking() || IsEpilogueTFEnabled)
     RUN_VPLAN_PASS(VPlanTransforms::foldTailByMasking, *VPlan0);
 
   RUN_VPLAN_PASS(VPlanTransforms::introduceMasksAndLinearize, *VPlan0);
@@ -6510,7 +6540,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
 }
 
 void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
-                                           ElementCount MaxVF) {
+                                           ElementCount MaxVF,
+                                           bool IsEpilogueTFEnabled) {
   if (ElementCount::isKnownGT(MinVF, MaxVF))
     return;
 
@@ -6542,6 +6573,8 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
       VPlans.push_back(std::move(P));
 
     TailFoldingStyle Style = CM->getTailFoldingStyle();
+    if (IsEpilogueTFEnabled)
+      Style = TailFoldingStyle::Data;
     RUN_VPLAN_PASS(VPlanTransforms::materializeHeaderMask, *Plan,
                    useActiveLaneMask(Style),
                    useActiveLaneMaskForControlFlow(Style));
@@ -7616,6 +7649,8 @@ fixScalarResumeValuesFromBypass(BasicBlock *BypassBlock, VPlan &BestEpiPlan,
     for (auto [ResumeV, HeaderPhi] :
          zip(ResumeValues, BestEpiPlan.getScalarHeader()->phis())) {
       auto *HeaderPhiR = cast<VPIRPhi>(&HeaderPhi);
+      if (!isa<PHINode>(HeaderPhiR->getIRPhi().getIncomingValueForBlock(PH)))
+        continue;
       auto *EpiResumePhi =
           cast<PHINode>(HeaderPhiR->getIRPhi().getIncomingValueForBlock(PH));
       if (EpiResumePhi->getBasicBlockIndex(BypassBlock) == -1)
@@ -7632,10 +7667,13 @@ fixScalarResumeValuesFromBypass(BasicBlock *BypassBlock, VPlan &BestEpiPlan,
 /// count check of the main loop, as well as updating various phis. \p
 /// InstsToMove contains instructions that need to be moved to the preheader of
 /// the epilogue vector loop.
-static void connectEpilogueVectorLoop(VPlan &EpiPlan, DominatorTree *DT,
+static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
+                                      DominatorTree *DT, LoopInfo *LI,
+                                      GeneratedRTChecks &Checks,
                                       VPIRBasicBlock *VecEpilogueIterCheckVPBB,
                                       ArrayRef<Instruction *> InstsToMove,
-                                      ArrayRef<VPInstruction *> ResumeValues) {
+                                      ArrayRef<VPInstruction *> ResumeValues,
+                                      bool IsEpilogueTFEnabled) {
   ArrayRef<VPBlockBase *> Preds = VecEpilogueIterCheckVPBB->getPredecessors();
   BasicBlock *MainLoopIterationCountCheck =
       cast<VPIRBasicBlock>(Preds.front())->getIRBasicBlock();
@@ -7653,6 +7691,31 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, DominatorTree *DT,
                     {DominatorTree::Insert, MainLoopIterationCountCheck,
                      VecEpiloguePreHeader}});
 
+  BasicBlock *ScalarPH =
+      cast<VPIRBasicBlock>(EpiPlan.getScalarPreheader())->getIRBasicBlock();
+  BasicBlock *SCEVCheckBlock = Checks.getSCEVChecks().second;
+  BasicBlock *MemCheckBlock = Checks.getMemRuntimeChecks().second;
+  // The epilogue plan's entry wraps the main loop's iteration count check
+  // (iter.check), which was redirected to the scalar preheader when executing
+  // the epilogue plan.
+  BasicBlock *EpilogueIterationCountCheck =
+      cast<VPIRBasicBlock>(EpiPlan.getEntry())->getIRBasicBlock();
+  if (IsEpilogueTFEnabled) {
+    // With a tail-folded epilogue there is no scalar remainder to bail
+    // to, even a trip count too small for the epilogue VF is handled safely by
+    // the masked epilogue vector loop, so skip straight to its preheader.
+    assert(is_contained(successors(EpilogueIterationCountCheck), ScalarPH) &&
+           "expected iter.check to branch to the scalar preheader");
+    ScalarPH->removePredecessor(EpilogueIterationCountCheck,
+                                /*KeepOneInputPHIs=*/true);
+    EpilogueIterationCountCheck->getTerminator()->replaceSuccessorWith(
+        ScalarPH, VecEpiloguePreHeader);
+    DTU.applyUpdates(
+        {{DominatorTree::Delete, EpilogueIterationCountCheck, ScalarPH},
+         {DominatorTree::Insert, EpilogueIterationCountCheck,
+          VecEpiloguePreHeader}});
+  }
+
   // The vec.epilog.iter.check block may contain Phi nodes from inductions
   // or reductions which merge control-flow from the latch block and the
   // middle block. Update the incoming values here and move the Phi into the
@@ -7665,6 +7728,18 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, DominatorTree *DT,
     Phi->replaceIncomingBlockWith(
         VecEpilogueIterationCountCheck->getSinglePredecessor(),
         VecEpilogueIterationCountCheck);
+    // When the epilogue is tail-folded, EpilogueIterationCountCheck
+    // (iter.check) is redirected to branch straight into the vector epilogue
+    // preheader (see the IsEpilogueTFEnabled redirect above), so it is now a
+    // genuine predecessor. Like MainLoopIterationCountCheck, it bypasses the
+    // main vector loop, so re-use the incoming value from that edge.
+    // TODO: revisit for reduction phis, whose resume value on this bypass
+    // edge may need dedicated handling rather than reusing the value already
+    // present here.
+    if (IsEpilogueTFEnabled)
+      Phi->addIncoming(
+          Phi->getIncomingValueForBlock(MainLoopIterationCountCheck),
+          EpilogueIterationCountCheck);
   }
 
   auto IP = VecEpiloguePreHeader->getFirstNonPHIIt();
@@ -7682,6 +7757,46 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, DominatorTree *DT,
   for (PHINode &Phi : make_early_inc_range(VecEpiloguePreHeader->phis()))
     if (Phi.use_empty())
       Phi.eraseFromParent();
+
+  if (IsEpilogueTFEnabled) {
+    // The epilogue vector loop is tail-folded, so it can safely handle
+    // any remaining trip count, including zero, via masking.
+    // vec.epilog.iter.check's own min-iters check was therefore built with a
+    // compile-time-known-false condition (see
+    // addMinimumVectorEpilogueIterationCheck) that never needs to bail out to
+    // a scalar remainder. Fold it into an unconditional branch into the
+    // vector epilogue preheader.
+    auto *Br =
+        cast<CondBrInst>(VecEpilogueIterationCountCheck->getTerminator());
+    [[maybe_unused]] auto *CondC = dyn_cast<ConstantInt>(Br->getCondition());
+    assert(CondC && CondC->isZero() &&
+           "expected vec.epilog.iter.check's branch condition to be a "
+           "compile-time false constant when the epilogue is tail-folded");
+    BasicBlock *DeadSucc = Br->getSuccessor(0);
+    UncondBrInst::Create(VecEpiloguePreHeader, Br->getIterator());
+    Br->eraseFromParent();
+    DTU.applyUpdates(
+        {{DominatorTree::Delete, VecEpilogueIterationCountCheck, DeadSucc}});
+  }
+
+  if (IsEpilogueTFEnabled && !SCEVCheckBlock && !MemCheckBlock) {
+    // No runtime check still needs a scalar fallback, and every trip-count
+    // bypass edge above has been redirected away from the scalar preheader:
+    // the scalar loop is now entirely unreachable. Delete it outright, along
+    // with its LoopInfo entry, instead of leaving it behind as dead code.
+    // This must run last, since fixScalarResumeValuesFromBypass (above) and
+    // the DT/function verification in processLoop (below) still need `L` and
+    // ScalarPH to be valid up to this point.
+    assert(pred_empty(ScalarPH) &&
+           "scalar preheader should have no predecessors left");
+    SmallVector<BasicBlock *> Blocks(L->block_begin(),
+                                     L->block_end());
+    Blocks.push_back(ScalarPH);
+    LI->erase(L);
+    for (auto *BB : Blocks)
+      LI->removeBlock(BB);
+    DeleteDeadBlocks(Blocks, &DTU);
+  }
 }
 
 bool LoopVectorizePass::processLoop(Loop *L) {
@@ -7878,14 +7993,12 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   EpilogueLowering EpilogueTailLoweringStatus =
       getEpilogueTailLowering(LVP.getCostModel(), L, ORE, LVL, Hints, TTI);
+  bool IsEpilogueTFEnabled = false;
   if (EpilogueTailLoweringStatus ==
       EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
     // TODO: Apply tail-folding on the vectorized epilogue loop.
-    LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is not supported yet\n");
-    reportVectorizationInfo(
-        "The epilogue-tail-folding policy prefer-fold-tail is not supported "
-        "yet, fall back to a normal epilogue",
-        "UnsupportedEpilogueTailFoldingPolicy", ORE, L);
+    LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
+    IsEpilogueTFEnabled = true;
   }
 
   // Get user vectorization factor and interleave count.
@@ -7899,8 +8012,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     UserIC = 1;
 
   // Plan how to best vectorize.
-  LVP.plan(UserVF, UserIC);
-  auto [VF, BestPlanPtr] = LVP.computeBestVF();
+  LVP.plan(UserVF, UserIC, IsEpilogueTFEnabled);
+  auto [VF, BestPlanPtr] = LVP.computeBestVF(IsEpilogueTFEnabled);
   unsigned IC = 1;
 
   // For VPlan build stress testing of outer loops, bail after plan
@@ -7916,7 +8029,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   GeneratedRTChecks Checks(PSE, DT, LI, TTI, Config.CostKind,
                            LVP.getCostModel().maskPartialAliasing());
-  if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width)) {
+  if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width,
+                                       LVP.getCostModel().foldTailByMasking())) {
     // Select the interleave count.
     IC = LVP.selectInterleaveCount(*BestPlanPtr, VF.Width, VF.Cost);
 
@@ -7978,7 +8092,9 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                   "Ignoring user-specified interleave count due to possibly "
                   "unsafe dependencies in the loop."};
     InterleaveLoop = false;
-  } else if (!LVP.hasPlanWithVF(VF.Width) && UserIC > 1) {
+  } else if (!LVP.hasPlanWithVF(VF.Width,
+                                LVP.getCostModel().foldTailByMasking()) &&
+             UserIC > 1) {
     // Tell the user interleaving was avoided up-front, despite being explicitly
     // requested.
     LLVM_DEBUG(dbgs() << "LV: Ignoring UserIC, because vectorization and "
@@ -8112,8 +8228,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   VPlan &BestPlan = *BestPlanPtr;
   // Consider vectorizing the epilogue too if it's profitable.
-  std::unique_ptr<VPlan> EpiPlan =
-      LVP.selectBestEpiloguePlan(BestPlan, VF.Width, IC, ScalarEpilogueAllowed);
+  std::unique_ptr<VPlan> EpiPlan = LVP.selectBestEpiloguePlan(
+      BestPlan, VF.Width, IC, ScalarEpilogueAllowed, IsEpilogueTFEnabled);
   bool HasBranchWeights =
       hasBranchWeightMD(*L->getLoopLatch()->getTerminator());
   if (EpiPlan) {
@@ -8145,6 +8261,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                                        BestMainPlan);
     auto ExpandedSCEVs = LVP.executePlan(
         EPI.MainLoopVF, EPI.MainLoopUF, BestMainPlan, MainILV, DT,
+        /*IsEpilogueTFEnabled*/ false,
         LoopVectorizationPlanner::EpilogueVectorizationKind::MainLoop);
     ++LoopsVectorized;
 
@@ -8162,10 +8279,11 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     RUN_VPLAN_PASS(VPlanTransforms::simplifyLiveInsWithSCEV, BestEpiPlan, PSE);
     LVP.executePlan(
         EPI.EpilogueVF, EPI.EpilogueUF, BestEpiPlan, EpilogILV, DT,
+        IsEpilogueTFEnabled,
         LoopVectorizationPlanner::EpilogueVectorizationKind::Epilogue);
-    connectEpilogueVectorLoop(BestEpiPlan, DT,
+    connectEpilogueVectorLoop(BestEpiPlan, L, DT, LI, Checks,
                               EpilogILV.VecEpilogueIterationCountCheck,
-                              InstsToMove, ResumeValues);
+                              InstsToMove, ResumeValues, IsEpilogueTFEnabled);
     ++LoopsEpilogueVectorized;
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
@@ -8177,7 +8295,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     if (!IsInnerLoop)
       LLVM_DEBUG(dbgs() << "Vectorizing outer loop in \"" << F->getName()
                         << "\"\n");
-    LVP.executePlan(VF.Width, IC, BestPlan, LB, DT);
+    LVP.executePlan(VF.Width, IC, BestPlan, LB, DT,
+                    /*IsEpilogueTFEnabled*/ false);
     ++LoopsVectorized;
   }
 
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.cpp b/llvm/lib/Transforms/Vectorize/VPlan.cpp
index 3cdad5161321a..df2c31528f114 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlan.cpp
@@ -1313,6 +1313,13 @@ VPIRBasicBlock *VPlan::createVPIRBasicBlock(BasicBlock *IRBB) {
   return VPIRBB;
 }
 
+bool VPlan::isCompatibleWithTF(bool TF) {
+  auto *VLR = getVectorLoopRegion();
+  assert(VLR && "Vector loop region got eliminated\n");
+  bool HasHeaderMask = (VLR->getHeaderMask() != nullptr);
+  return HasHeaderMask == TF;
+}
+
 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
 
 Twine VPlanPrinter::getUID(const VPBlockBase *Block) {
@@ -1645,19 +1652,29 @@ bool LoopVectorizationPlanner::getDecisionAndClampRange(
   return PredicateAtRangeStart;
 }
 
-VPlan &LoopVectorizationPlanner::getPlanFor(ElementCount VF) const {
+VPlan &LoopVectorizationPlanner::getPlanFor(ElementCount VF, bool TF) const {
   assert(count_if(VPlans,
-                  [VF](const VPlanPtr &Plan) { return Plan->hasVF(VF); }) ==
-             1 &&
+                  [VF, TF](const VPlanPtr &Plan) {
+                    LLVM_DEBUG(dbgs() << "LV: given VF: " << VF << " and TF: "
+                                      << TF << " equivalent vplan: ";
+                               Plan->dump());
+                    return Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF);
+                  }) == 1 &&
          "Multiple VPlans for VF.");
 
   for (const VPlanPtr &Plan : VPlans) {
-    if (Plan->hasVF(VF))
+    if (Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF))
       return *Plan.get();
   }
   llvm_unreachable("No plan found!");
 }
 
+bool LoopVectorizationPlanner::hasPlanWithVF(ElementCount VF, bool TF) const {
+  return any_of(VPlans, [VF, TF](const VPlanPtr &Plan) {
+    return Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF);
+  });
+}
+
 static void addRuntimeUnrollDisableMetaData(Loop *L) {
   SmallVector<Metadata *, 4> MDs;
   // Reserve first location for self reference to the LoopID metadata node.
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 746c0231f6fcf..e31cde1587811 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -5258,6 +5258,15 @@ class VPlan {
            is_contained(ScalarPH->getPredecessors(), getMiddleBlock());
   }
 
+  /// Returns true if this VPlan's tail-folding matches \p TF.
+  /// A plan is tail-folded if it has a header mask, or if it has no scalar
+  /// tail while the TC is not evenly divisible by VF (in which case the tail
+  /// must be folded by the vector loop itself).
+  ///
+  /// This function is used to disambiguate VPlan lookup by VF when multiple
+  /// plans share the same VF but differ in tail-folding status.
+  bool isCompatibleWithTF(bool TF);
+
   /// The type of the canonical induction variable of the vector loop.
   Type *getIndexType() const { return VF.getType(); }
 };
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 6c5f688835c15..6be809d1d1f05 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1620,9 +1620,25 @@ void VPlanTransforms::addMinimumVectorEpilogueIterationCheck(
     ElementCount EpilogueVF, unsigned EpilogueUF, unsigned MainLoopStep,
     unsigned EpilogueLoopStep, ScalarEvolution &SE) {
   // Add the minimum iteration check for the epilogue vector loop.
+  VPBuilder Builder(cast<VPBasicBlock>(Plan.getEntry()));
+
+  if (Plan.isCompatibleWithTF(/*TF*/ true)) {
+    Builder.createNaryOp(VPInstruction::BranchOnCond, Plan.getFalse());
+    return;
+  }
+  // if (Plan.isCompatibleWithTF(/*TF*/ true)) {
+  //   VPBasicBlock *EntryVPBB = cast<VPBasicBlock>(Plan.getEntry());
+  //   // Successor 0 is the "taken" (scalar) edge, successor 1 is the vector
+  //   // path — see BranchOnCond's execute() and removeBranchOnConst's
+  //   // RemovedIdx convention. Drop the scalar edge directly, leaving Entry
+  //   // with a single successor (which needs no BranchOnCond terminator at
+  //   // all — VPlan codegen emits a plain unconditional `br` for that case).
+  //   VPBlockUtils::disconnectBlocks(EntryVPBB, EntryVPBB->getSuccessors()[0]);
+  //   return;
+  // }
+
   VPValue *TC = Plan.getTripCount();
   Value *TripCount = TC->getLiveInIRValue();
-  VPBuilder Builder(cast<VPBasicBlock>(Plan.getEntry()));
   VPValue *VFxUF = Builder.createExpandSCEV(SE.getElementCount(
       TripCount->getType(), (EpilogueVF * EpilogueUF), SCEV::FlagNUW));
   VPValue *Count = Builder.createSub(TC, Plan.getOrAddLiveIn(VectorTripCount),
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
new file mode 100644
index 0000000000000..495b280990aec
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -0,0 +1,63 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail -mtriple=aarch64-linux-gnu -S %s | FileCheck %s
+
+
+define void @test_epilogue_tf(ptr %A, i64 %n) {
+; CHECK-LABEL: define void @test_epilogue_tf(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = and i64 [[N]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 16
+; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP0]], align 1
+; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP1]], align 1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[N_RND_UP:%.*]] = add i64 [[N]], 7
+; CHECK-NEXT:    [[N_MOD_VF2:%.*]] = and i64 [[N_RND_UP]], 7
+; CHECK-NEXT:    [[N_VEC3:%.*]] = sub i64 [[N_RND_UP]], [[N_MOD_VF2]]
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT5:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX4]]
+; CHECK-NEXT:    store <8 x i8> splat (i8 1), ptr [[TMP3]], align 1
+; CHECK-NEXT:    [[INDEX_NEXT5]] = add nuw i64 [[INDEX4]], 8
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT5]], [[N_VEC3]]
+; CHECK-NEXT:    br i1 [[TMP4]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
+  store i8 1, ptr %arrayidx, align 1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond = icmp ne i64 %iv.next, %n
+  br i1 %exitcond, label %for.body, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index 1b3964fe1978d..a48e8bcc6d2e2 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -28,8 +28,7 @@
 
 define void @test_epilogue_tf(ptr %A, i64 %n, i8 %val) {
 ; CHECK-LABEL: LV: Checking a loop in 'test_epilogue_tf'
-; CHECK: LV: epilogue tail-folding is not supported yet
-; CHECK: remark: <unknown>:0:0: The epilogue-tail-folding policy prefer-fold-tail is not supported yet, fall back to a normal epilogue
+; CHECK: LV: epilogue tail-folding is enabled
 ;
 ; CHECK-DISABLED-EPILOG-LABEL: LV: Checking a loop in 'test_epilogue_tf'
 ; CHECK-DISABLED-EPILOG: remark: <unknown>:0:0: Options conflict, epilogue vectorization is disallowed while epilogue tail-folding allowed!

>From b11534aa1dbfa9e35b55d9f8204e7bb078e15152 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Tue, 21 Jul 2026 15:10:26 +0100
Subject: [PATCH 02/25] rebase and resolve review comments - limit the feature
 to be enabled only for forced epilogue VF

---
 .../Vectorize/LoopVectorizationPlanner.h      | 10 +--
 .../Transforms/Vectorize/LoopVectorize.cpp    | 65 ++++++-------------
 llvm/lib/Transforms/Vectorize/VPlan.cpp       | 25 ++-----
 llvm/lib/Transforms/Vectorize/VPlan.h         |  9 ---
 .../Vectorize/VPlanConstruction.cpp           | 12 +---
 .../AArch64/fold-epilogue-tail.ll             |  3 +-
 6 files changed, 33 insertions(+), 91 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 1f69de6d42e53..d3fe00dc6b687 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -952,13 +952,12 @@ class LoopVectorizationPlanner {
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
-  VPlan &getPlanFor(ElementCount VF, bool TF) const;
+  VPlan &getPlanFor(ElementCount VF) const;
 
   /// Compute and return the most profitable vectorization factor and the
   /// corresponding best VPlan. Also collect all profitable VFs in
   /// ProfitableVFs.
-  std::pair<VectorizationFactor, VPlan *>
-  computeBestVF(bool IsEpilogueTFEnabled);
+  std::pair<VectorizationFactor, VPlan *> computeBestVF();
 
   /// \return The desired interleave count.
   /// If interleave count has been specified by metadata it will be returned.
@@ -994,7 +993,10 @@ class LoopVectorizationPlanner {
 
   /// Look through the existing plans and return true if we have one with
   /// vectorization factor \p VF.
-  bool hasPlanWithVF(ElementCount VF, bool TF) const;
+  bool hasPlanWithVF(ElementCount VF) const {
+    return any_of(VPlans,
+                  [&](const VPlanPtr &Plan) { return Plan->hasVF(VF); });
+  }
 
   /// Test a \p Predicate on a \p Range of VF's. Return the value of applying
   /// \p Predicate on Range.Start, possibly decreasing Range.End such that the
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index e2f57c97af55c..700bf04d5f39a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3164,8 +3164,6 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
   using RecipeVFPair = std::pair<VPRecipeBase *, ElementCount>;
   SmallVector<RecipeVFPair> InvalidCosts;
   for (const auto &Plan : VPlans) {
-    if (!Plan->isCompatibleWithTF(CM->foldTailByMasking()))
-      continue;
     for (ElementCount VF : Plan->vectorFactors()) {
       // The VPlan-based cost model is designed for computing vector cost.
       // Querying VPlan-based cost model with a scarlar VF will cause some
@@ -3554,12 +3552,9 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     }
 
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization factor is forced.\n");
-    if (hasPlanWithVF(EpilogueVectorizationForceVF,
-                      MainPlan.hasTailFolded() || IsEpilogueTFEnabled)) {
+    if (hasPlanWithVF(EpilogueVectorizationForceVF)) {
       std::unique_ptr<VPlan> Clone(
-          getPlanFor(EpilogueVectorizationForceVF,
-                    MainPlan.hasTailFolded() || IsEpilogueTFEnabled)
-              .duplicate());
+          getPlanFor(EpilogueVectorizationForceVF).duplicate());
       Clone->setVF(EpilogueVectorizationForceVF);
       return Clone;
     }
@@ -3650,12 +3645,10 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
   VPlan *BestPlan = nullptr;
   for (auto &NextVF : ProfitableVFs) {
     // Skip candidate VFs without a corresponding VPlan.
-    if (!hasPlanWithVF(NextVF.Width,
-                       MainPlan.hasTailFolded() || IsEpilogueTFEnabled))
+    if (!hasPlanWithVF(NextVF.Width))
       continue;
 
-    VPlan &CurrentPlan = getPlanFor(
-        NextVF.Width, MainPlan.hasTailFolded() || IsEpilogueTFEnabled);
+    VPlan &CurrentPlan = getPlanFor(NextVF.Width);
     ElementCount EffectiveVF = GetEffectiveVF(CurrentPlan, NextVF.Width);
     // Skip fixed vector VFs > than the estimated runtime VF, or any VF > than
     // the VF of the main loop.
@@ -5418,8 +5411,13 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
         CM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF,
-                    /*IsEpilogueTFEnabled*/ false);
+        if (IsEpilogueTFEnabled) {
+          auto TFVPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ true);
+          buildVPlans(*TFVPlan1, EpilogueUserVF, EpilogueUserVF,
+                      /*IsEpilogueTFEnabled*/ true);
+        } else
+          buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF,
+                      /*IsEpilogueTFEnabled*/ false);
       }
       if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
@@ -5456,22 +5454,6 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
   buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
               /*IsEpilogueTFEnabled*/ false);
   LLVM_DEBUG(printPlans(dbgs()));
-
-  // Build tail-folded vplans when IsEpilogueTFEnabled is enabled:
-  if (IsEpilogueTFEnabled) {
-    auto TFVPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ true);
-    if (!TFVPlan1)
-      return;
-    buildVPlans(*TFVPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF,
-                /*IsEpilogueTFEnabled*/ true);
-    buildVPlans(*TFVPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
-                /*IsEpilogueTFEnabled*/ true);
-    LLVM_DEBUG(dbgs() << "LV: Tail-folded vplans:\n");
-    for (auto &vplan : VPlans) {
-      if (vplan->isCompatibleWithTF(true))
-        LLVM_DEBUG(vplan->dump());
-    }
-  }
 }
 
 VPCostContext::VPCostContext(const TargetLibraryInfo &TLI, const VPlan &Plan,
@@ -5655,7 +5637,7 @@ InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
 }
 
 std::pair<VectorizationFactor, VPlan *>
-LoopVectorizationPlanner::computeBestVF(bool IsEpilogueTFEnabled) {
+LoopVectorizationPlanner::computeBestVF() {
   if (VPlans.empty())
     return {VectorizationFactor::Disabled(), nullptr};
   // If there is a single VPlan with a single VF, return it directly.
@@ -5665,15 +5647,13 @@ LoopVectorizationPlanner::computeBestVF(bool IsEpilogueTFEnabled) {
   if (VPlans.size() == 1) {
     // For outer loops, the plan has a single vector VF determined by the
     // heuristic.
-    assert((FirstPlan.hasScalarVFOnly() ||
-            hasPlanWithVF(UserVF, CM->foldTailByMasking()) ||
+    assert((FirstPlan.hasScalarVFOnly() || hasPlanWithVF(UserVF) ||
             FirstPlan.isOuterLoop()) &&
            "must have a single scalar VF, UserVF or an outer loop");
     return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
   }
 
-  if (hasPlanWithVF(UserVF, CM->foldTailByMasking()) && hasForcedEpilogueVF() &&
-      VPlans.size() == 2) {
+  if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF() && VPlans.size() == 2) {
     assert(VPlans[0]->getSingleVF() == UserVF &&
            "expected second plan to be for the forced UserVF");
     assert(VPlans[1]->getSingleVF() == EpilogueVectorizationForceVF &&
@@ -5758,11 +5738,9 @@ LoopVectorizationPlanner::computeBestVF(bool IsEpilogueTFEnabled) {
           cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
-      if (P->isCompatibleWithTF(CM->foldTailByMasking())) {
-        if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
-          BestFactor = CurrentFactor;
-          PlanForBestVF = P.get();
-        }
+      if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
+        BestFactor = CurrentFactor;
+        PlanForBestVF = P.get();
       }
 
       // If profitable add it to ProfitableVF list.
@@ -8013,7 +7991,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC, IsEpilogueTFEnabled);
-  auto [VF, BestPlanPtr] = LVP.computeBestVF(IsEpilogueTFEnabled);
+  auto [VF, BestPlanPtr] = LVP.computeBestVF();
   unsigned IC = 1;
 
   // For VPlan build stress testing of outer loops, bail after plan
@@ -8029,8 +8007,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   GeneratedRTChecks Checks(PSE, DT, LI, TTI, Config.CostKind,
                            LVP.getCostModel().maskPartialAliasing());
-  if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width,
-                                       LVP.getCostModel().foldTailByMasking())) {
+  if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width)) {
     // Select the interleave count.
     IC = LVP.selectInterleaveCount(*BestPlanPtr, VF.Width, VF.Cost);
 
@@ -8092,9 +8069,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                   "Ignoring user-specified interleave count due to possibly "
                   "unsafe dependencies in the loop."};
     InterleaveLoop = false;
-  } else if (!LVP.hasPlanWithVF(VF.Width,
-                                LVP.getCostModel().foldTailByMasking()) &&
-             UserIC > 1) {
+  } else if (!LVP.hasPlanWithVF(VF.Width) && UserIC > 1) {
     // Tell the user interleaving was avoided up-front, despite being explicitly
     // requested.
     LLVM_DEBUG(dbgs() << "LV: Ignoring UserIC, because vectorization and "
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.cpp b/llvm/lib/Transforms/Vectorize/VPlan.cpp
index df2c31528f114..3cdad5161321a 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlan.cpp
@@ -1313,13 +1313,6 @@ VPIRBasicBlock *VPlan::createVPIRBasicBlock(BasicBlock *IRBB) {
   return VPIRBB;
 }
 
-bool VPlan::isCompatibleWithTF(bool TF) {
-  auto *VLR = getVectorLoopRegion();
-  assert(VLR && "Vector loop region got eliminated\n");
-  bool HasHeaderMask = (VLR->getHeaderMask() != nullptr);
-  return HasHeaderMask == TF;
-}
-
 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
 
 Twine VPlanPrinter::getUID(const VPBlockBase *Block) {
@@ -1652,29 +1645,19 @@ bool LoopVectorizationPlanner::getDecisionAndClampRange(
   return PredicateAtRangeStart;
 }
 
-VPlan &LoopVectorizationPlanner::getPlanFor(ElementCount VF, bool TF) const {
+VPlan &LoopVectorizationPlanner::getPlanFor(ElementCount VF) const {
   assert(count_if(VPlans,
-                  [VF, TF](const VPlanPtr &Plan) {
-                    LLVM_DEBUG(dbgs() << "LV: given VF: " << VF << " and TF: "
-                                      << TF << " equivalent vplan: ";
-                               Plan->dump());
-                    return Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF);
-                  }) == 1 &&
+                  [VF](const VPlanPtr &Plan) { return Plan->hasVF(VF); }) ==
+             1 &&
          "Multiple VPlans for VF.");
 
   for (const VPlanPtr &Plan : VPlans) {
-    if (Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF))
+    if (Plan->hasVF(VF))
       return *Plan.get();
   }
   llvm_unreachable("No plan found!");
 }
 
-bool LoopVectorizationPlanner::hasPlanWithVF(ElementCount VF, bool TF) const {
-  return any_of(VPlans, [VF, TF](const VPlanPtr &Plan) {
-    return Plan->hasVF(VF) && Plan->isCompatibleWithTF(TF);
-  });
-}
-
 static void addRuntimeUnrollDisableMetaData(Loop *L) {
   SmallVector<Metadata *, 4> MDs;
   // Reserve first location for self reference to the LoopID metadata node.
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index e31cde1587811..746c0231f6fcf 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -5258,15 +5258,6 @@ class VPlan {
            is_contained(ScalarPH->getPredecessors(), getMiddleBlock());
   }
 
-  /// Returns true if this VPlan's tail-folding matches \p TF.
-  /// A plan is tail-folded if it has a header mask, or if it has no scalar
-  /// tail while the TC is not evenly divisible by VF (in which case the tail
-  /// must be folded by the vector loop itself).
-  ///
-  /// This function is used to disambiguate VPlan lookup by VF when multiple
-  /// plans share the same VF but differ in tail-folding status.
-  bool isCompatibleWithTF(bool TF);
-
   /// The type of the canonical induction variable of the vector loop.
   Type *getIndexType() const { return VF.getType(); }
 };
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 6be809d1d1f05..92b5801d50269 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1622,20 +1622,10 @@ void VPlanTransforms::addMinimumVectorEpilogueIterationCheck(
   // Add the minimum iteration check for the epilogue vector loop.
   VPBuilder Builder(cast<VPBasicBlock>(Plan.getEntry()));
 
-  if (Plan.isCompatibleWithTF(/*TF*/ true)) {
+  if (Plan.hasTailFolded()) {
     Builder.createNaryOp(VPInstruction::BranchOnCond, Plan.getFalse());
     return;
   }
-  // if (Plan.isCompatibleWithTF(/*TF*/ true)) {
-  //   VPBasicBlock *EntryVPBB = cast<VPBasicBlock>(Plan.getEntry());
-  //   // Successor 0 is the "taken" (scalar) edge, successor 1 is the vector
-  //   // path — see BranchOnCond's execute() and removeBranchOnConst's
-  //   // RemovedIdx convention. Drop the scalar edge directly, leaving Entry
-  //   // with a single successor (which needs no BranchOnCond terminator at
-  //   // all — VPlan codegen emits a plain unconditional `br` for that case).
-  //   VPBlockUtils::disconnectBlocks(EntryVPBB, EntryVPBB->getSuccessors()[0]);
-  //   return;
-  // }
 
   VPValue *TC = Plan.getTripCount();
   Value *TripCount = TC->getLiveInIRValue();
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 495b280990aec..a94cf64668ff6 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,5 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail -mtriple=aarch64-linux-gnu -S %s | FileCheck %s
+; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -mtriple=aarch64-linux-gnu -S %s | FileCheck %s
 
 
 define void @test_epilogue_tf(ptr %A, i64 %n) {

>From e063dffb97f3c6c7169c7d76c2e9824a9422976b Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Thu, 23 Jul 2026 11:57:23 +0100
Subject: [PATCH 03/25] Enable correct costs for tail-folded epilogue. To
 achieve that, I pass the epilogue CM to cost/vplan-related functions.

---
 .../Vectorize/LoopVectorizationPlanner.h      |  25 +-
 .../Transforms/Vectorize/LoopVectorize.cpp    | 232 ++++++++++++------
 .../AArch64/fold-epilogue-tail.ll             |  37 +--
 3 files changed, 190 insertions(+), 104 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index d3fe00dc6b687..3d25dc64e68fe 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -917,7 +917,8 @@ class LoopVectorizationPlanner {
   ///
   /// TODO: Move to VPlan::cost once the use of LoopVectorizationLegality has
   /// been retired.
-  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU) const;
+  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU,
+                       LoopVectorizationCostModel &EnabledCM) const;
 
   /// Precompute costs for certain instructions using the legacy cost model. The
   /// function is used to bring up the VPlan-based cost model to initially avoid
@@ -948,7 +949,14 @@ class LoopVectorizationPlanner {
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// interleaving should be avoided up-front, no plans are generated.
-  void plan(ElementCount UserVF, unsigned UserIC, bool IsEpilogueTFEnabled);
+  void plan(ElementCount UserVF, unsigned UserIC);
+
+  /// Build VPlans for the specified \p EpilogueUserVF and \p IC if they are
+  /// non-zero or all applicable candidate VFs otherwise. If vectorization and
+  /// tail-folding should be avoided up-front, no plans are generated.
+  bool planForEpilogueTF(ElementCount UserVF, unsigned UserIC,
+                         ElementCount EpilogueUserVF,
+                         LoopVectorizationCostModel &EpilogueCM);
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
@@ -983,7 +991,6 @@ class LoopVectorizationPlanner {
   DenseMap<const SCEV *, Value *>
   executePlan(ElementCount VF, unsigned UF, VPlan &BestPlan,
               InnerLoopVectorizer &LB, DominatorTree *DT,
-              bool IsEpilogueTFEnabled,
               EpilogueVectorizationKind EpilogueVecKind =
                   EpilogueVectorizationKind::None);
 
@@ -1010,13 +1017,10 @@ class LoopVectorizationPlanner {
   /// Returns nullptr if epilogue vectorization is not supported or not
   /// profitable for the loop. \p ScalarEpilogueAllowed indicates whether the
   /// epilogue lowering policy permits creating a scalar epilogue at all.
-  /// \p IsEpilogueTFEnabled indicates whether the epilogue loop should be
-  /// tail-folded rather than left with a scalar epilogue of its own.
   std::unique_ptr<VPlan> selectBestEpiloguePlan(VPlan &MainPlan,
                                                 ElementCount MainLoopVF,
                                                 unsigned IC,
-                                                bool ScalarEpilogueAllowed,
-                                                bool IsEpilogueTFEnabled);
+                                                bool ScalarEpilogueAllowed);
 
   /// Emit remarks for recipes with invalid costs in the available VPlans.
   void emitInvalidCostRemarks(OptimizationRemarkEmitter *ORE);
@@ -1050,7 +1054,7 @@ class LoopVectorizationPlanner {
   /// Build an initial VPlan, with HCFG wrapping the original scalar loop and
   /// scalar transformations applied. Returns null if an initial VPlan cannot
   /// be built.
-  VPlanPtr tryToBuildVPlan1(bool IsEpilogueTFEnabled);
+  VPlanPtr tryToBuildVPlan1(LoopVectorizationCostModel &EnabledCM);
 
   /// Build a VPlan using VPRecipes according to the information gathered by
   /// Legal and VPlan-based analysis. For outer loops, performs basic recipe
@@ -1060,13 +1064,14 @@ class LoopVectorizationPlanner {
   /// maximum VF for which no plan could be built. Each VPlan is built starting
   /// from a copy of \p InitialPlan, which is a plain CFG VPlan wrapping the
   /// original scalar loop.
-  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range);
+  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range,
+                           LoopVectorizationCostModel &EnabledCM);
 
   /// Build VPlans for power-of-2 VF's between \p MinVF and \p MaxVF inclusive,
   /// based on \p VPlan1 and according to the information gathered by Legal
   /// when it checked if it is legal to vectorize the loop.
   void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
-                   bool IsEpilogueTFEnabled);
+                   LoopVectorizationCostModel &EnabledCM);
 
   /// Add ComputeReductionResult recipes to the middle block to compute the
   /// final reduction results. Add Select recipes to the latch block when
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 700bf04d5f39a..fe975db56b90c 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3512,7 +3512,7 @@ bool VFSelectionContext::isEpilogueVectorizationProfitable(
 
 std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     VPlan &MainPlan, ElementCount MainLoopVF, unsigned IC,
-    bool ScalarEpilogueAllowed, bool IsEpilogueTFEnabled) {
+    bool ScalarEpilogueAllowed) {
   if (!EnableEpilogueVectorization) {
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization is disabled.\n");
     return nullptr;
@@ -3749,7 +3749,7 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
     if (VF.isScalar())
       LoopCost = CM->expectedCost(VF);
     else
-      LoopCost = cost(Plan, VF, &R);
+      LoopCost = cost(Plan, VF, &R, *CM);
     assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
 
     // Loop body is free and there is no need for interleaving.
@@ -5342,8 +5342,7 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
   }
 }
 
-void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
-                                    bool IsEpilogueTFEnabled) {
+void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   CM->collectValuesToIgnore();
   Config.collectElementTypesForWidening(&CM->ValuesToIgnore);
 
@@ -5358,7 +5357,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
   if (MaxFactors.FixedVF.isVector() || MaxFactors.ScalableVF.isVector())
     Legal->collectUnitStridePredicates();
 
-  auto VPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ false);
+  auto VPlan1 = tryToBuildVPlan1(*CM);
   if (!VPlan1)
     return;
 
@@ -5367,7 +5366,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
     // plan for that VF only.
     ElementCount VF =
         MaxFactors.FixedVF ? MaxFactors.FixedVF : MaxFactors.ScalableVF;
-    buildVPlans(*VPlan1, VF, VF, /*IsEpilogueTFEnabled*/ false);
+    buildVPlans(*VPlan1, VF, VF, *CM);
     LLVM_DEBUG(printPlans(dbgs()));
     return;
   }
@@ -5406,24 +5405,19 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
       CM->collectNonVectorizedAndSetWideningDecisions(UserVF);
-      buildVPlans(*VPlan1, UserVF, UserVF, /*IsEpilogueTFEnabled*/ false);
+      buildVPlans(*VPlan1, UserVF, UserVF, *CM);
+
       ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
         CM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        if (IsEpilogueTFEnabled) {
-          auto TFVPlan1 = tryToBuildVPlan1(/*IsEpilogueTFEnabled*/ true);
-          buildVPlans(*TFVPlan1, EpilogueUserVF, EpilogueUserVF,
-                      /*IsEpilogueTFEnabled*/ true);
-        } else
-          buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF,
-                      /*IsEpilogueTFEnabled*/ false);
+        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, *CM);
       }
       if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
         // vector VFs only.
         if (UserVF.isScalar() ||
-            cost(*VPlans.front(), UserVF, /*RU=*/nullptr).isValid()) {
+            cost(*VPlans.front(), UserVF, /*RU=*/nullptr, *CM).isValid()) {
           LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
           LLVM_DEBUG(printPlans(dbgs()));
           return;
@@ -5449,13 +5443,74 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC,
     CM->collectNonVectorizedAndSetWideningDecisions(VF);
   }
 
-  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF,
-              /*IsEpilogueTFEnabled*/ false);
+  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF, *CM);
   buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
-              /*IsEpilogueTFEnabled*/ false);
+              *CM);
+
   LLVM_DEBUG(printPlans(dbgs()));
 }
 
+bool LoopVectorizationPlanner::planForEpilogueTF(
+    ElementCount UserVF, unsigned UserIC, ElementCount EpilogueUserVF,
+    LoopVectorizationCostModel &EpilogueCM) {
+  if (VPlans.empty())
+    return false;
+  if (!OrigLoop->isInnermost())
+    return false;
+
+  if (!EpilogueUserVF.isVector() ||
+      ElementCount::isKnownGE(EpilogueUserVF, UserVF))
+    return false;
+
+  EpilogueCM.ValuesToIgnore.insert_range(CM->ValuesToIgnore);
+  EpilogueCM.VecValuesToIgnore.insert_range(CM->VecValuesToIgnore);
+
+  FixedScalableVFPair MaxFactors =
+      EpilogueCM.computeMaxVF(EpilogueUserVF, UserIC);
+  if (!MaxFactors ||
+      !EpilogueCM.foldTailByMasking()) // Cases that should not to be vectorized
+                                       // nor tail-folded.
+    return false;
+
+  auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
+  if (!VPlan1)
+    return false;
+
+  // Invalidate interleave groups if all blocks of loop will be predicated.
+  if (EpilogueCM.blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
+      !useMaskedInterleavedAccesses(TTI)) {
+    LLVM_DEBUG(
+        dbgs() << "LV: [EpilogueTF] Invalidate all interleaved groups due to "
+                  "fold-tail "
+                  "by masking which requires masked-interleaved support.\n");
+    if (EpilogueCM.InterleaveInfo.invalidateGroups())
+      // Invalidating interleave groups also requires invalidating all decisions
+      // based on them, which includes widening decisions and uniform and scalar
+      // values.
+      EpilogueCM.invalidateCostModelingDecisions();
+  }
+
+  if (EpilogueCM.foldTailByMasking())
+    Legal->prepareToFoldTailByMasking();
+
+  // Collect the instructions (and their associated costs) that will be more
+  // profitable to scalarize.
+  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
+
+  assert(VPlans.size() == 2 &&
+         "For tail-folded epilogue, VPlans size is expected to be 2");
+  // remove last vplan which should be the epilogue plan to replace it by our
+  // tail-folded vplan:
+  assert(VPlans.back()->getSingleVF() == EpilogueUserVF &&
+         "For tail-folded epilogue, first vplan is expected to have "
+         "EpilogueUserVF");
+  VPlans.pop_back();
+  buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, EpilogueCM);
+
+  cost(*VPlans.back(), EpilogueUserVF, /*RU=*/nullptr, EpilogueCM);
+  return true;
+}
+
 VPCostContext::VPCostContext(const TargetLibraryInfo &TLI, const VPlan &Plan,
                              LoopVectorizationCostModel &CM,
                              VFSelectionContext &Config,
@@ -5599,9 +5654,10 @@ getRecordedExecutionFrequency(const VPBasicBlock *VPBB) {
 }
 #endif
 
-InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
-                                               VPRegisterUsage *RU) const {
-  VPCostContext CostCtx(*TLI, Plan, *CM, Config,
+InstructionCost LoopVectorizationPlanner::cost(
+    VPlan &Plan, ElementCount VF, VPRegisterUsage *RU,
+    LoopVectorizationCostModel &EnabledCM) const {
+  VPCostContext CostCtx(*TLI, Plan, EnabledCM, Config,
                         /*ReusePrintingSlotTracker=*/true);
   InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
 
@@ -5735,7 +5791,7 @@ LoopVectorizationPlanner::computeBestVF() {
       }
 
       InstructionCost Cost =
-          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
+          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr, *CM);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
       if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
@@ -5775,7 +5831,7 @@ void LoopVectorizationPlanner::clearCostModel() { CM.reset(); }
 
 DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
     ElementCount BestVF, unsigned BestUF, VPlan &BestVPlan,
-    InnerLoopVectorizer &ILV, DominatorTree *DT, bool IsEpilogueTFEnabled,
+    InnerLoopVectorizer &ILV, DominatorTree *DT,
     EpilogueVectorizationKind EpilogueVecKind) {
   assert(BestVPlan.hasVF(BestVF) &&
          "Trying to execute plan with unsupported VF");
@@ -6420,7 +6476,7 @@ static bool verifyExecutionFrequenciesMatchBFI(VPlan &Plan, Loop *OrigLoop,
 }
 #endif
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(bool IsEpilogueTFEnabled) {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(LoopVectorizationCostModel &EnabledCM) {
   bool IsInnerLoop = OrigLoop->isInnermost();
 
   // Set up loop versioning for inner loops with memory runtime checks.
@@ -6476,8 +6532,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(bool IsEpilogueTFEnabled) {
       Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled;
   bool OptForSize =
       !ForceVectorization &&
-      (CM->EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
-       CM->EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
+      (EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
+       EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
   unsigned SCEVCheckThreshold = ForceVectorization
                                     ? PragmaVectorizeSCEVCheckThreshold
                                     : VectorizeSCEVCheckThreshold;
@@ -6509,7 +6565,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(bool IsEpilogueTFEnabled) {
 
   RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
                  getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
-  if (CM->foldTailByMasking() || IsEpilogueTFEnabled)
+  if (EnabledCM.foldTailByMasking())
     RUN_VPLAN_PASS(VPlanTransforms::foldTailByMasking, *VPlan0);
 
   RUN_VPLAN_PASS(VPlanTransforms::introduceMasksAndLinearize, *VPlan0);
@@ -6517,17 +6573,17 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(bool IsEpilogueTFEnabled) {
   return VPlan0;
 }
 
-void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
-                                           ElementCount MaxVF,
-                                           bool IsEpilogueTFEnabled) {
+void LoopVectorizationPlanner::buildVPlans(
+    VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
+    LoopVectorizationCostModel &EnabledCM) {
   if (ElementCount::isKnownGT(MinVF, MaxVF))
     return;
 
   auto MaxVFTimes2 = MaxVF * 2;
   for (ElementCount VF = MinVF; ElementCount::isKnownLT(VF, MaxVFTimes2);) {
     VFRange SubRange = {VF, MaxVFTimes2};
-    auto Plan =
-        tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()), SubRange);
+    auto Plan = tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()),
+                                SubRange, EnabledCM);
     VF = SubRange.End;
 
     if (!Plan)
@@ -6540,7 +6596,7 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
                    Config.getMinimalBitwidths());
     RUN_VPLAN_PASS(VPlanTransforms::optimize, *Plan);
     // TODO: try to put addExplicitVectorLength close to addActiveLaneMask
-    if (CM->foldTailWithEVL()) {
+    if (EnabledCM.foldTailWithEVL()) {
       RUN_VPLAN_PASS(VPlanTransforms::addExplicitVectorLength, *Plan,
                      Config.getMaxSafeElements());
       RUN_VPLAN_PASS(VPlanTransforms::optimizeEVLMasks, *Plan);
@@ -6550,9 +6606,7 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
             RUN_VPLAN_PASS(VPlanTransforms::narrowInterleaveGroups, *Plan, TTI))
       VPlans.push_back(std::move(P));
 
-    TailFoldingStyle Style = CM->getTailFoldingStyle();
-    if (IsEpilogueTFEnabled)
-      Style = TailFoldingStyle::Data;
+    TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
     RUN_VPLAN_PASS(VPlanTransforms::materializeHeaderMask, *Plan,
                    useActiveLaneMask(Style),
                    useActiveLaneMaskForControlFlow(Style));
@@ -6563,8 +6617,8 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
   }
 }
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
-                                                   VFRange &Range) {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
+    VPlanPtr Plan, VFRange &Range, LoopVectorizationCostModel &EnabledCM) {
 
   // For outer loops, the plan only needs basic recipe conversion and induction
   // live-out optimization; the full inner-loop recipe building below does not
@@ -6590,8 +6644,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
 
   bool RequiresScalarEpilogueCheck =
       LoopVectorizationPlanner::getDecisionAndClampRange(
-          [this](ElementCount VF) {
-            return !CM->requiresScalarEpilogue(VF.isVector());
+          [&EnabledCM](ElementCount VF) {
+            return !EnabledCM.requiresScalarEpilogue(VF.isVector());
           },
           Range);
   // Update the branch in the middle block if a scalar epilogue is required.
@@ -6609,9 +6663,9 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // TODO: Consider using getDecisionAndClampRange here to split up VPlans.
   bool IVUpdateMayOverflow = false;
   for (ElementCount VF : Range)
-    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(CM.get(), VF);
+    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(&EnabledCM, VF);
 
-  TailFoldingStyle Style = CM->getTailFoldingStyle();
+  TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
   // Use NUW for the induction increment if we proved that it won't overflow in
   // the vector loop or when not folding the tail. In the later case, we know
   // that the canonical induction increment will not overflow as the vector trip
@@ -6637,10 +6691,11 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // Range, add it to the set of groups to be later applied to the VPlan and add
   // placeholders for its members' Recipes which we'll be replacing with a
   // single VPInterleaveRecipe.
-  for (InterleaveGroup<Instruction> *IG : IAI.getInterleaveGroups()) {
-    auto ApplyIG = [IG, this](ElementCount VF) -> bool {
+  for (InterleaveGroup<Instruction> *IG :
+       EnabledCM.InterleaveInfo.getInterleaveGroups()) {
+    auto ApplyIG = [IG, &EnabledCM](ElementCount VF) -> bool {
       bool Result = (VF.isVector() && // Query is illegal for VF == 1
-                     CM->getWideningDecision(IG->getInsertPos(), VF) ==
+                     EnabledCM.getWideningDecision(IG->getInsertPos(), VF) ==
                          LoopVectorizationCostModel::CM_Interleave);
       // For scalable vectors, the interleave factors must be <= 8 since we
       // require the (de)interleaveN intrinsics instead of shufflevectors.
@@ -6657,12 +6712,12 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // Construct wide recipes and apply predication for original scalar
   // VPInstructions in the loop.
   // ---------------------------------------------------------------------------
-  VPRecipeBuilder RecipeBuilder(*Plan, Legal, *CM, Builder);
+  VPRecipeBuilder RecipeBuilder(*Plan, Legal, EnabledCM, Builder);
 
   RUN_VPLAN_PASS(VPlanTransforms::createInLoopReductionRecipes, *Plan,
                  Range.Start);
 
-  VPCostContext CostCtx(*TLI, *Plan, *CM, Config);
+  VPCostContext CostCtx(*TLI, *Plan, EnabledCM, Config);
 
   RUN_VPLAN_PASS(VPlanTransforms::makeMemOpWideningDecisions, *Plan, Range,
                  RecipeBuilder, CostCtx);
@@ -6769,7 +6824,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // for this VPlan, replace the Recipes widening its memory instructions with a
   // single VPInterleaveRecipe at its insertion point.
   RUN_VPLAN_PASS(VPlanTransforms::createInterleaveGroups, *Plan,
-                 InterleaveGroups, CM->isEpilogueAllowed());
+                 InterleaveGroups, EnabledCM.isEpilogueAllowed());
 
   // Convert memory recipes to strided access recipes if the strided access is
   // legal and profitable.
@@ -6786,7 +6841,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
 
   RUN_VPLAN_PASS(VPlanTransforms::dropPoisonGeneratingRecipes, *Plan);
 
-  if (CM->maskPartialAliasing())
+  if (EnabledCM.maskPartialAliasing())
     RUN_VPLAN_PASS(VPlanTransforms::attachAliasMaskToHeaderMask, *Plan);
 
   assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
@@ -7436,7 +7491,8 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
   VPValue *VPV = Plan.getOrAddLiveIn(EPResumeVal);
   assert(all_of(IV->users(),
                 [](const VPUser *U) {
-                  if (isa<VPScalarIVStepsRecipe, VPDerivedIVRecipe>(U))
+                  if (isa<VPScalarIVStepsRecipe, VPDerivedIVRecipe,
+                          VPWidenCanonicalIVRecipe>(U))
                     return true;
                   unsigned Opc = cast<VPInstruction>(U)->getOpcode();
                   return Instruction::isCast(Opc) || Opc == Instruction::Add;
@@ -7450,7 +7506,7 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
   auto *Increment = vputils::findCanonicalIVIncrement(Plan);
   assert(Increment && "Must have a canonical IV increment at this point");
   IV->replaceUsesWithIf(Add, [Add, Increment](VPUser &U, unsigned) {
-    return &U != Add && &U != Increment;
+    return &U != Add && &U != Increment && !isa<VPWidenCanonicalIVRecipe>(&U);
   });
   VPInstruction *OffsetIVInc =
       VPBuilder::getToInsertAfter(Increment).createAdd(Increment, VPV);
@@ -7561,6 +7617,23 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
           continue;
         }
       }
+    } else if (isa<VPActiveLaneMaskPHIRecipe>(&R)) {
+      // The active-lane-mask phi's entry value was computed assuming the
+      // epilogue vector loop starts at 0. Rebuild it using the resume value
+      // instead, so the mask reflects how many elements the main vector loop
+      // already processed.
+      VPBuilder EntryBuilder(Plan.getVectorPreheader());
+      Type *CanIVTy = VectorLoop->getCanonicalIVType();
+      VPValue *ALMMultiplier = Plan.getConstantInt(CanIVTy, 1);
+      auto *EntryIncrement = EntryBuilder.createOverflowingOp(
+          VPInstruction::CanonicalIVIncrementForPart, {VPV, &Plan.getVF()}, {},
+          R.getDebugLoc(), "index.part.next");
+      auto *EntryALM = EntryBuilder.createNaryOp(
+          VPInstruction::ActiveLaneMask,
+          {EntryIncrement, Plan.getTripCount(), ALMMultiplier}, R.getDebugLoc(),
+          "active.lane.mask.entry");
+      cast<VPHeaderPHIRecipe>(&R)->setStartValue(EntryALM);
+      continue;
     } else {
       // Retrieve the induction resume value via ResumeForEpilogue.
       PHINode *IndPhi = cast<VPWidenInductionRecipe>(&R)->getPHINode();
@@ -7755,25 +7828,18 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
     Br->eraseFromParent();
     DTU.applyUpdates(
         {{DominatorTree::Delete, VecEpilogueIterationCountCheck, DeadSucc}});
-  }
 
-  if (IsEpilogueTFEnabled && !SCEVCheckBlock && !MemCheckBlock) {
-    // No runtime check still needs a scalar fallback, and every trip-count
-    // bypass edge above has been redirected away from the scalar preheader:
-    // the scalar loop is now entirely unreachable. Delete it outright, along
-    // with its LoopInfo entry, instead of leaving it behind as dead code.
-    // This must run last, since fixScalarResumeValuesFromBypass (above) and
-    // the DT/function verification in processLoop (below) still need `L` and
-    // ScalarPH to be valid up to this point.
-    assert(pred_empty(ScalarPH) &&
-           "scalar preheader should have no predecessors left");
-    SmallVector<BasicBlock *> Blocks(L->block_begin(),
-                                     L->block_end());
-    Blocks.push_back(ScalarPH);
-    LI->erase(L);
-    for (auto *BB : Blocks)
-      LI->removeBlock(BB);
-    DeleteDeadBlocks(Blocks, &DTU);
+    if (!SCEVCheckBlock && !MemCheckBlock) {
+      // Delete the scalar loop as it's dead right
+      assert(pred_empty(ScalarPH) &&
+             "scalar preheader should have no predecessors left");
+      SmallVector<BasicBlock *> Blocks(L->block_begin(), L->block_end());
+      Blocks.push_back(ScalarPH);
+      LI->erase(L);
+      for (auto *BB : Blocks)
+        LI->removeBlock(BB);
+      DeleteDeadBlocks(Blocks, &DTU);
+    }
   }
 }
 
@@ -7972,11 +8038,19 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   EpilogueLowering EpilogueTailLoweringStatus =
       getEpilogueTailLowering(LVP.getCostModel(), L, ORE, LVL, Hints, TTI);
   bool IsEpilogueTFEnabled = false;
+  std::optional<InterleavedAccessInfo> TailFoldingCMIAI;
+  std::optional<LoopVectorizationCostModel> EpilogueTailFoldingCM;
   if (EpilogueTailLoweringStatus ==
       EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
-    // TODO: Apply tail-folding on the vectorized epilogue loop.
     LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
     IsEpilogueTFEnabled = true;
+    TailFoldingCMIAI.emplace(PSE, L, DT, LI, LVL.getLAI(), OptForSize);
+    if (UseInterleaved && useMaskedInterleavedAccesses(*TTI))
+      TailFoldingCMIAI->analyzeInterleaving(
+          /*useMaskedInterleavedAccesses*/ true);
+    EpilogueTailFoldingCM.emplace(CM_EpilogueNotNeededFoldTail, L, PSE, LI,
+                                  &LVL, *TTI, TLI, AC, ORE, GetBFI, F,
+                                  *TailFoldingCMIAI, Config);
   }
 
   // Get user vectorization factor and interleave count.
@@ -7990,7 +8064,16 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     UserIC = 1;
 
   // Plan how to best vectorize.
-  LVP.plan(UserVF, UserIC, IsEpilogueTFEnabled);
+  LVP.plan(UserVF, UserIC);
+  if (IsEpilogueTFEnabled)
+    if (!LVP.planForEpilogueTF(UserVF, /*UserIC*/ 1,
+                               EpilogueVectorizationForceVF,
+                               *EpilogueTailFoldingCM)) {
+      // we can't apply epilogue TF:
+      EpilogueTailFoldingCM.reset();
+      IsEpilogueTFEnabled = false;
+    }
+
   auto [VF, BestPlanPtr] = LVP.computeBestVF();
   unsigned IC = 1;
 
@@ -8204,7 +8287,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   VPlan &BestPlan = *BestPlanPtr;
   // Consider vectorizing the epilogue too if it's profitable.
   std::unique_ptr<VPlan> EpiPlan = LVP.selectBestEpiloguePlan(
-      BestPlan, VF.Width, IC, ScalarEpilogueAllowed, IsEpilogueTFEnabled);
+      BestPlan, VF.Width, IC, ScalarEpilogueAllowed);
   bool HasBranchWeights =
       hasBranchWeightMD(*L->getLoopLatch()->getTerminator());
   if (EpiPlan) {
@@ -8236,7 +8319,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                                        BestMainPlan);
     auto ExpandedSCEVs = LVP.executePlan(
         EPI.MainLoopVF, EPI.MainLoopUF, BestMainPlan, MainILV, DT,
-        /*IsEpilogueTFEnabled*/ false,
         LoopVectorizationPlanner::EpilogueVectorizationKind::MainLoop);
     ++LoopsVectorized;
 
@@ -8254,7 +8336,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     RUN_VPLAN_PASS(VPlanTransforms::simplifyLiveInsWithSCEV, BestEpiPlan, PSE);
     LVP.executePlan(
         EPI.EpilogueVF, EPI.EpilogueUF, BestEpiPlan, EpilogILV, DT,
-        IsEpilogueTFEnabled,
         LoopVectorizationPlanner::EpilogueVectorizationKind::Epilogue);
     connectEpilogueVectorLoop(BestEpiPlan, L, DT, LI, Checks,
                               EpilogILV.VecEpilogueIterationCountCheck,
@@ -8270,8 +8351,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     if (!IsInnerLoop)
       LLVM_DEBUG(dbgs() << "Vectorizing outer loop in \"" << F->getName()
                         << "\"\n");
-    LVP.executePlan(VF.Width, IC, BestPlan, LB, DT,
-                    /*IsEpilogueTFEnabled*/ false);
+    LVP.executePlan(VF.Width, IC, BestPlan, LB, DT);
     ++LoopsVectorized;
   }
 
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index a94cf64668ff6..746e4495e059e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,11 +1,11 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -mtriple=aarch64-linux-gnu -S %s | FileCheck %s
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug -mtriple=aarch64-linux-gnu -mcpu=neoverse-v1 -S %s | FileCheck %s
 
 
 define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK-LABEL: define void @test_epilogue_tf(
-; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ITER_CHECK:.*]]:
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
@@ -13,18 +13,18 @@ define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[N_MOD_VF:%.*]] = and i64 [[N]], 31
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[TMP0]], i64 16
-; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP0]], align 1
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP1]], i64 16
 ; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP1]], align 1
+; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP2]], align 1
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
@@ -32,17 +32,18 @@ define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[N_RND_UP:%.*]] = add i64 [[N]], 7
-; CHECK-NEXT:    [[N_MOD_VF2:%.*]] = and i64 [[N_RND_UP]], 7
-; CHECK-NEXT:    [[N_VEC3:%.*]] = sub i64 [[N_RND_UP]], [[N_MOD_VF2]]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT5:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX4]]
-; CHECK-NEXT:    store <8 x i8> splat (i8 1), ptr [[TMP3]], align 1
-; CHECK-NEXT:    [[INDEX_NEXT5]] = add nuw i64 [[INDEX4]], 8
-; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT5]], [[N_VEC3]]
-; CHECK-NEXT:    br i1 [[TMP4]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX2]]
+; CHECK-NEXT:    call void @llvm.masked.store.v8i8.p0(<8 x i8> splat (i8 1), ptr align 1 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP6:%.*]] = xor i1 [[TMP5]], true
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    br label %[[EXIT]]
 ; CHECK:       [[EXIT]]:

>From 8e04f864e48677fd6cc21532048195fe1fd8d493 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Thu, 30 Jul 2026 17:41:23 +0100
Subject: [PATCH 04/25] resolve review comments

---
 .../Transforms/Vectorize/LoopVectorize.cpp    | 66 ++++++++++---------
 .../Vectorize/VPlanConstruction.cpp           |  2 +
 .../AArch64/fold-epilogue-tail.ll             | 33 +++++++++-
 .../LoopVectorize/fold-epilogue-tail.ll       | 56 +++++++++++++++-
 4 files changed, 125 insertions(+), 32 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index fe975db56b90c..ff095d7059f46 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5405,6 +5405,9 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
       CM->collectNonVectorizedAndSetWideningDecisions(UserVF);
+      // Build the main-loop VPlan firstly because if epilogue tail-folding is
+      // enabled, it will be built later, so we keep the epilogue vplans at the
+      // end.
       buildVPlans(*VPlan1, UserVF, UserVF, *CM);
 
       ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
@@ -5453,56 +5456,61 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 bool LoopVectorizationPlanner::planForEpilogueTF(
     ElementCount UserVF, unsigned UserIC, ElementCount EpilogueUserVF,
     LoopVectorizationCostModel &EpilogueCM) {
-  if (VPlans.empty())
+  if (VPlans.empty()) {
+    LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
+                         "out of epilogue tail-folding\n");
     return false;
+  }
   if (!OrigLoop->isInnermost())
     return false;
 
   if (!EpilogueUserVF.isVector() ||
-      ElementCount::isKnownGE(EpilogueUserVF, UserVF))
+      (estimateElementCount(EpilogueUserVF, Config.getVScaleForTuning()) >=
+       estimateElementCount(UserVF, Config.getVScaleForTuning())) *
+          UserIC)
     return false;
 
   EpilogueCM.ValuesToIgnore.insert_range(CM->ValuesToIgnore);
   EpilogueCM.VecValuesToIgnore.insert_range(CM->VecValuesToIgnore);
 
   FixedScalableVFPair MaxFactors =
-      EpilogueCM.computeMaxVF(EpilogueUserVF, UserIC);
+      EpilogueCM.computeMaxVF(EpilogueUserVF, /*UserIC*/ 1);
   if (!MaxFactors ||
-      !EpilogueCM.foldTailByMasking()) // Cases that should not to be vectorized
-                                       // nor tail-folded.
+      !EpilogueCM
+           .foldTailByMasking()) { // Cases that should not to be vectorized
+                                   // or tail-folded.
+    reportVectorizationInfo("This case of epilogue loop can't be tail-folded",
+                            "InvalidTailFoldedEpilogue", ORE, OrigLoop);
     return false;
+  }
 
   auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
   if (!VPlan1)
     return false;
 
-  // Invalidate interleave groups if all blocks of loop will be predicated.
-  if (EpilogueCM.blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
-      !useMaskedInterleavedAccesses(TTI)) {
+  if (!useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
-        dbgs() << "LV: [EpilogueTF] Invalidate all interleaved groups due to "
-                  "fold-tail "
-                  "by masking which requires masked-interleaved support.\n");
+        dbgs() << "LV: Invalidate all interleaved groups due to fold-tail by "
+                  "masking which requires masked-interleaved support.\n");
     if (EpilogueCM.InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
       EpilogueCM.invalidateCostModelingDecisions();
   }
-
-  if (EpilogueCM.foldTailByMasking())
-    Legal->prepareToFoldTailByMasking();
+  Legal->prepareToFoldTailByMasking();
 
   // Collect the instructions (and their associated costs) that will be more
   // profitable to scalarize.
   EpilogueCM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
 
+  // Expecting only 2 vplans as epilogue TF is applied only for forced VFs.
   assert(VPlans.size() == 2 &&
          "For tail-folded epilogue, VPlans size is expected to be 2");
-  // remove last vplan which should be the epilogue plan to replace it by our
-  // tail-folded vplan:
+  // Remove the last vplan, which should be the epilogue plan to replace it by
+  // the tail-folded vplan:
   assert(VPlans.back()->getSingleVF() == EpilogueUserVF &&
-         "For tail-folded epilogue, first vplan is expected to have "
+         "For tail-folded epilogue, last vplan is expected to have "
          "EpilogueUserVF");
   VPlans.pop_back();
   buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, EpilogueCM);
@@ -7811,7 +7819,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
 
   if (IsEpilogueTFEnabled) {
     // The epilogue vector loop is tail-folded, so it can safely handle
-    // any remaining trip count, including zero, via masking.
+    // any remaining iterations, including zero, via masking.
     // vec.epilog.iter.check's own min-iters check was therefore built with a
     // compile-time-known-false condition (see
     // addMinimumVectorEpilogueIterationCheck) that never needs to bail out to
@@ -7830,7 +7838,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
         {{DominatorTree::Delete, VecEpilogueIterationCountCheck, DeadSucc}});
 
     if (!SCEVCheckBlock && !MemCheckBlock) {
-      // Delete the scalar loop as it's dead right
+      // Delete the scalar loop as it's dead right now.
       assert(pred_empty(ScalarPH) &&
              "scalar preheader should have no predecessors left");
       SmallVector<BasicBlock *> Blocks(L->block_begin(), L->block_end());
@@ -8037,17 +8045,14 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   EpilogueLowering EpilogueTailLoweringStatus =
       getEpilogueTailLowering(LVP.getCostModel(), L, ORE, LVL, Hints, TTI);
-  bool IsEpilogueTFEnabled = false;
   std::optional<InterleavedAccessInfo> TailFoldingCMIAI;
   std::optional<LoopVectorizationCostModel> EpilogueTailFoldingCM;
   if (EpilogueTailLoweringStatus ==
       EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
     LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
-    IsEpilogueTFEnabled = true;
     TailFoldingCMIAI.emplace(PSE, L, DT, LI, LVL.getLAI(), OptForSize);
-    if (UseInterleaved && useMaskedInterleavedAccesses(*TTI))
-      TailFoldingCMIAI->analyzeInterleaving(
-          /*useMaskedInterleavedAccesses*/ true);
+    if (UseInterleaved)
+      TailFoldingCMIAI->analyzeInterleaving(useMaskedInterleavedAccesses(*TTI));
     EpilogueTailFoldingCM.emplace(CM_EpilogueNotNeededFoldTail, L, PSE, LI,
                                   &LVL, *TTI, TLI, AC, ORE, GetBFI, F,
                                   *TailFoldingCMIAI, Config);
@@ -8065,13 +8070,13 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC);
-  if (IsEpilogueTFEnabled)
-    if (!LVP.planForEpilogueTF(UserVF, /*UserIC*/ 1,
-                               EpilogueVectorizationForceVF,
-                               *EpilogueTailFoldingCM)) {
+  if (EpilogueTailFoldingCM.has_value())
+    if (!LVP.planForEpilogueTF(UserVF, UserIC, EpilogueVectorizationForceVF,
+                               EpilogueTailFoldingCM.value())) {
       // we can't apply epilogue TF:
+      LLVM_DEBUG(
+          dbgs() << "LV: Applying epilogue tail-folding failed, disable it.\n");
       EpilogueTailFoldingCM.reset();
-      IsEpilogueTFEnabled = false;
     }
 
   auto [VF, BestPlanPtr] = LVP.computeBestVF();
@@ -8339,7 +8344,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
         LoopVectorizationPlanner::EpilogueVectorizationKind::Epilogue);
     connectEpilogueVectorLoop(BestEpiPlan, L, DT, LI, Checks,
                               EpilogILV.VecEpilogueIterationCountCheck,
-                              InstsToMove, ResumeValues, IsEpilogueTFEnabled);
+                              InstsToMove, ResumeValues,
+                              EpilogueTailFoldingCM.has_value());
     ++LoopsEpilogueVectorized;
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 92b5801d50269..02f71df83f57e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1623,6 +1623,8 @@ void VPlanTransforms::addMinimumVectorEpilogueIterationCheck(
   VPBuilder Builder(cast<VPBasicBlock>(Plan.getEntry()));
 
   if (Plan.hasTailFolded()) {
+    assert(!RequiresScalarEpilogue &&
+           "Expected no scalar epilogue for tail-folded plan");
     Builder.createNaryOp(VPInstruction::BranchOnCond, Plan.getFalse());
     return;
   }
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 746e4495e059e..02ce78dcc8e14 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,7 +1,11 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug -mtriple=aarch64-linux-gnu -mcpu=neoverse-v1 -S %s | FileCheck %s
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug -mcpu=neoverse-v1 -S %s | FileCheck %s
 
+; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALIDATE-INTERLEAVE
+
+target triple = "aarch64-linux-gnu"
 
 define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK-LABEL: define void @test_epilogue_tf(
@@ -63,3 +67,30 @@ for.body:
 exit:
   ret void
 }
+
+define i64 @test_no_masked_interleave_support(i64 %y, i32 %n) {
+; CHECK-INVALIDATE-INTERLEAVE-LABEL: Checking a loop in 'test_no_masked_interleave_support'
+; CHECK-INVALIDATE-INTERLEAVE: LV: epilogue tail-folding is enabled
+; CHECK-INVALIDATE-INTERLEAVE: LV: Analyzing interleaved accesses...
+; CHECK-INVALIDATE-INTERLEAVE: LV: Invalidate all interleaved groups due to fold-tail by masking which requires masked-interleaved support
+entry:
+  br label %for.body
+
+for.body:
+  %i = phi i32 [ 0, %entry ], [ %inc, %cond.end ]
+  %cmp = icmp eq i64 %y, 0
+  br i1 %cmp, label %cond.end, label %cond.false
+
+cond.false:
+  %div = xor i64 3, %y
+  br label %cond.end
+
+cond.end:
+  %cond = phi i64 [ %div, %cond.false ], [ 77, %for.body ]
+  %inc = add nuw nsw i32 %i, 1
+  %exitcond = icmp eq i32 %inc, %n
+  br i1 %exitcond, label %for.cond.cleanup, label %for.body
+
+for.cond.cleanup:
+  ret i64 %cond
+}
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index a48e8bcc6d2e2..3ec2204034bf9 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -26,6 +26,11 @@
 ; RUN: %{cmd} -force-vector-width=16 -epilogue-vectorization-force-VF=8 \
 ; RUN: -enable-interleaved-mem-accesses=true < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALID-INTERLEAVE
 
+; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -vectorize-scev-check-threshold=0 < %s 2>&1 | FileCheck %s \
+; RUN: --check-prefix=CHECK-NO-VPLANS
+
+
 define void @test_epilogue_tf(ptr %A, i64 %n, i8 %val) {
 ; CHECK-LABEL: LV: Checking a loop in 'test_epilogue_tf'
 ; CHECK: LV: epilogue tail-folding is enabled
@@ -60,6 +65,29 @@ exit:
   ret void
 }
 
+; This case can't be tail-folded because all the iterations will be executed by
+; main vector loop.
+define void @test_epilogue_tf_reset(ptr %A) {
+; CHECK-LABEL: LV: Checking a loop in 'test_epilogue_tf_reset'
+; CHECK: LV: epilogue tail-folding is enabled
+; CHECK: LV: This case of epilogue loop can't be tail-folded.
+; CHECK: LV: Applying epilogue tail-folding failed, disable it.
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
+  store i8 1, ptr %arrayidx, align 1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond = icmp ne i64 %iv.next, 64
+  br i1 %exitcond, label %for.body, label %exit
+
+exit:
+  ret void
+}
+
 define i16 @require_scalar_epilogue(ptr %dst, i64 %x) {
 ; CHECK-LABEL: Checking a loop in 'require_scalar_epilogue'
 ; CHECK: remark: <unknown>:0:0: Epilogue tail-folding can't be applied because scalar epilogue is required. Fall back to a normal epilogue
@@ -277,6 +305,32 @@ exit:
   ret void
 }
 
+; Can't build a valid vplan for this case because too many SCEV checks needed,
+; more than the specfied limit.
+define i64 @test_no_vplan_built(ptr %dst, i64 %n) {
+; CHECK-NO-VPLANS-LABEL: Checking a loop in 'test_no_vplan_built'
+; CHECK-NO-VPLANS: LV: epilogue tail-folding is enabled
+; CHECK-NO-VPLANS: LV: no vplans have been built for main loop VF, bail out of epilogue tail-folding
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %dead.iv = phi i16 [ 0, %entry ], [ %dead.iv.next, %loop ]
+  %prev = phi i64 [ 0, %entry ], [ %ext, %loop ]
+  %iv.next = add nuw nsw i64 %iv, 1
+  %dead.iv.next = add i16 %dead.iv, 1
+  %ext = zext i16 %dead.iv.next to i64
+  %gep = getelementptr inbounds i64, ptr %dst, i64 %prev
+  store i64 %iv, ptr %gep, align 8
+  %cmp = icmp slt i64 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  %result = phi i64 [ %ext, %loop ]
+  ret i64 %result
+}
+
 !1 = distinct !{!1, !2}
 !2 = !{!"llvm.loop.vectorize.enable"}
-

>From c3963cce842301512d130d2b6c16d3ef67bc9a59 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Thu, 6 Aug 2026 00:43:59 +0100
Subject: [PATCH 05/25] resolve review comments

---
 .../Vectorize/LoopVectorizationPlanner.h      |  1 -
 .../Transforms/Vectorize/LoopVectorize.cpp    | 35 ++++++++-----------
 .../AArch64/fold-epilogue-tail.ll             |  5 +--
 .../LoopVectorize/fold-epilogue-tail.ll       |  6 ++--
 4 files changed, 20 insertions(+), 27 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 3d25dc64e68fe..d4309f44adf78 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -955,7 +955,6 @@ class LoopVectorizationPlanner {
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// tail-folding should be avoided up-front, no plans are generated.
   bool planForEpilogueTF(ElementCount UserVF, unsigned UserIC,
-                         ElementCount EpilogueUserVF,
                          LoopVectorizationCostModel &EpilogueCM);
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ff095d7059f46..7375af1b01477 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5454,27 +5454,19 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 }
 
 bool LoopVectorizationPlanner::planForEpilogueTF(
-    ElementCount UserVF, unsigned UserIC, ElementCount EpilogueUserVF,
+    ElementCount UserVF, unsigned UserIC,
     LoopVectorizationCostModel &EpilogueCM) {
   if (VPlans.empty()) {
     LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
                          "out of epilogue tail-folding\n");
     return false;
   }
-  if (!OrigLoop->isInnermost())
-    return false;
-
-  if (!EpilogueUserVF.isVector() ||
-      (estimateElementCount(EpilogueUserVF, Config.getVScaleForTuning()) >=
-       estimateElementCount(UserVF, Config.getVScaleForTuning())) *
-          UserIC)
-    return false;
 
   EpilogueCM.ValuesToIgnore.insert_range(CM->ValuesToIgnore);
   EpilogueCM.VecValuesToIgnore.insert_range(CM->VecValuesToIgnore);
 
   FixedScalableVFPair MaxFactors =
-      EpilogueCM.computeMaxVF(EpilogueUserVF, /*UserIC*/ 1);
+      EpilogueCM.computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
   if (!MaxFactors ||
       !EpilogueCM
            .foldTailByMasking()) { // Cases that should not to be vectorized
@@ -5485,8 +5477,6 @@ bool LoopVectorizationPlanner::planForEpilogueTF(
   }
 
   auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
-  if (!VPlan1)
-    return false;
 
   if (!useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
@@ -5502,20 +5492,23 @@ bool LoopVectorizationPlanner::planForEpilogueTF(
 
   // Collect the instructions (and their associated costs) that will be more
   // profitable to scalarize.
-  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
+  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(
+      EpilogueVectorizationForceVF);
 
   // Expecting only 2 vplans as epilogue TF is applied only for forced VFs.
   assert(VPlans.size() == 2 &&
          "For tail-folded epilogue, VPlans size is expected to be 2");
   // Remove the last vplan, which should be the epilogue plan to replace it by
   // the tail-folded vplan:
-  assert(VPlans.back()->getSingleVF() == EpilogueUserVF &&
+  assert(VPlans.back()->getSingleVF() == EpilogueVectorizationForceVF &&
          "For tail-folded epilogue, last vplan is expected to have "
          "EpilogueUserVF");
   VPlans.pop_back();
-  buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, EpilogueCM);
+  buildVPlans(*VPlan1, EpilogueVectorizationForceVF,
+              EpilogueVectorizationForceVF, EpilogueCM);
 
-  cost(*VPlans.back(), EpilogueUserVF, /*RU=*/nullptr, EpilogueCM);
+  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr,
+       EpilogueCM);
   return true;
 }
 
@@ -8070,12 +8063,12 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC);
-  if (EpilogueTailFoldingCM.has_value())
-    if (!LVP.planForEpilogueTF(UserVF, UserIC, EpilogueVectorizationForceVF,
-                               EpilogueTailFoldingCM.value())) {
+  if (EpilogueTailFoldingCM)
+    if (!LVP.planForEpilogueTF(UserVF, UserIC, EpilogueTailFoldingCM.value())) {
       // we can't apply epilogue TF:
-      LLVM_DEBUG(
-          dbgs() << "LV: Applying epilogue tail-folding failed, disable it.\n");
+      reportVectorizationInfo(
+          "Applying epilogue tail-folding failed, disable it.",
+          "InvalidTailFoldedEpilogue", ORE, L);
       EpilogueTailFoldingCM.reset();
     }
 
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 02ce78dcc8e14..e9b248016c32f 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,8 +1,9 @@
+; REQUIRES: asserts
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug -mcpu=neoverse-v1 -S %s | FileCheck %s
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug-only=loop-vectorize -mcpu=neoverse-v1 -S %s | FileCheck %s
 
-; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize,vectorutils --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALIDATE-INTERLEAVE
 
 target triple = "aarch64-linux-gnu"
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index 3ec2204034bf9..c4a7097411e3c 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -26,7 +26,7 @@
 ; RUN: %{cmd} -force-vector-width=16 -epilogue-vectorization-force-VF=8 \
 ; RUN: -enable-interleaved-mem-accesses=true < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALID-INTERLEAVE
 
-; RUN: opt -S -p loop-vectorize -debug --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -vectorize-scev-check-threshold=0 < %s 2>&1 | FileCheck %s \
 ; RUN: --check-prefix=CHECK-NO-VPLANS
 
@@ -67,8 +67,8 @@ exit:
 
 ; This case can't be tail-folded because all the iterations will be executed by
 ; main vector loop.
-define void @test_epilogue_tf_reset(ptr %A) {
-; CHECK-LABEL: LV: Checking a loop in 'test_epilogue_tf_reset'
+define void @test_no_iterations_left(ptr %A) {
+; CHECK-LABEL: LV: Checking a loop in 'test_no_iterations_left'
 ; CHECK: LV: epilogue tail-folding is enabled
 ; CHECK: LV: This case of epilogue loop can't be tail-folded.
 ; CHECK: LV: Applying epilogue tail-folding failed, disable it.

>From 531a89cf42571582d2fd7a89923e5a5374869f4f Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Sun, 9 Aug 2026 00:45:24 +0100
Subject: [PATCH 06/25] Swap between default CM instance and EPilogueTF one to
 enable costs for tail-folded epilogue

---
 .../Vectorize/LoopVectorizationPlanner.h      |  25 ++-
 .../Transforms/Vectorize/LoopVectorize.cpp    | 182 +++++++++---------
 .../AArch64/fold-epilogue-tail.ll             |   2 +-
 3 files changed, 107 insertions(+), 102 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index d4309f44adf78..7d3fc5e52e076 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -887,7 +887,8 @@ class LoopVectorizationPlanner {
 
   /// The profitability analysis. Cleared after making cost based decisions.
   std::unique_ptr<LoopVectorizationCostModel> CM;
-
+  LoopVectorizationCostModel *EnabledCM;
+  LoopVectorizationCostModel *EpilogueTFCM;
   /// VF selection state independent of cost-modeling decisions.
   VFSelectionContext &Config;
 
@@ -917,8 +918,7 @@ class LoopVectorizationPlanner {
   ///
   /// TODO: Move to VPlan::cost once the use of LoopVectorizationLegality has
   /// been retired.
-  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU,
-                       LoopVectorizationCostModel &EnabledCM) const;
+  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU) const;
 
   /// Precompute costs for certain instructions using the legacy cost model. The
   /// function is used to bring up the VPlan-based cost model to initially avoid
@@ -931,6 +931,7 @@ class LoopVectorizationPlanner {
       Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
       const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
       std::unique_ptr<LoopVectorizationCostModel> CM,
+      LoopVectorizationCostModel *EpilogueTFCM,
       VFSelectionContext &Config, InterleavedAccessInfo &IAI,
       PredicatedScalarEvolution &PSE, OptimizationRemarkEmitter *ORE,
       std::function<const BranchProbabilityInfo &()> GetBPI);
@@ -946,6 +947,13 @@ class LoopVectorizationPlanner {
   /// Destroy the cost model.
   void clearCostModel();
 
+  void enableDefaultCM() { EnabledCM = &*CM; }
+
+  void enableEpilogueTFCM() {
+    assert(EpilogueTFCM && "No CM for epilogue tail-folding to enable");
+    EnabledCM = EpilogueTFCM;
+  }
+
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// interleaving should be avoided up-front, no plans are generated.
@@ -954,8 +962,7 @@ class LoopVectorizationPlanner {
   /// Build VPlans for the specified \p EpilogueUserVF and \p IC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// tail-folding should be avoided up-front, no plans are generated.
-  bool planForEpilogueTF(ElementCount UserVF, unsigned UserIC,
-                         LoopVectorizationCostModel &EpilogueCM);
+  bool planForEpilogueTF(ElementCount UserVF, unsigned UserIC);
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
@@ -1053,7 +1060,7 @@ class LoopVectorizationPlanner {
   /// Build an initial VPlan, with HCFG wrapping the original scalar loop and
   /// scalar transformations applied. Returns null if an initial VPlan cannot
   /// be built.
-  VPlanPtr tryToBuildVPlan1(LoopVectorizationCostModel &EnabledCM);
+  VPlanPtr tryToBuildVPlan1();
 
   /// Build a VPlan using VPRecipes according to the information gathered by
   /// Legal and VPlan-based analysis. For outer loops, performs basic recipe
@@ -1063,14 +1070,12 @@ class LoopVectorizationPlanner {
   /// maximum VF for which no plan could be built. Each VPlan is built starting
   /// from a copy of \p InitialPlan, which is a plain CFG VPlan wrapping the
   /// original scalar loop.
-  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range,
-                           LoopVectorizationCostModel &EnabledCM);
+  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range);
 
   /// Build VPlans for power-of-2 VF's between \p MinVF and \p MaxVF inclusive,
   /// based on \p VPlan1 and according to the information gathered by Legal
   /// when it checked if it is legal to vectorize the loop.
-  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
-                   LoopVectorizationCostModel &EnabledCM);
+  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF);
 
   /// Add ComputeReductionResult recipes to the middle block to compute the
   /// final reduction results. Add Select recipes to the latch block when
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 7375af1b01477..9de61f2f09af8 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3172,7 +3172,7 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
       if (VF.isScalar())
         continue;
 
-      VPCostContext CostCtx(*TLI, *Plan, *CM, Config,
+      VPCostContext CostCtx(*TLI, *Plan, *EnabledCM, Config,
                             /*ReusePrintingSlotTracker=*/true);
       precomputeCosts(*Plan, VF, CostCtx);
       auto Iter = vp_depth_first_deep(Plan->getVectorLoopRegion()->getEntry());
@@ -3713,7 +3713,7 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   // Do not interleave tail-folded loops, as the overhead of multiple
   // instructions to calculate the predicate is likely not beneficial.
   // If an epilogue is not allowed for any other reason, do not interleave.
-  if (!CM->isEpilogueAllowed())
+  if (!EnabledCM->isEpilogueAllowed())
     return 1;
 
   if (any_of(Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis(),
@@ -3747,9 +3747,9 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   // then we calculate the cost of VF here.
   if (LoopCost == 0) {
     if (VF.isScalar())
-      LoopCost = CM->expectedCost(VF);
+      LoopCost = EnabledCM->expectedCost(VF);
     else
-      LoopCost = cost(Plan, VF, &R, *CM);
+      LoopCost = cost(Plan, VF, &R);
     assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
 
     // Loop body is free and there is no need for interleaving.
@@ -3830,7 +3830,7 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   auto BestKnownTC =
       getSmallBestKnownTC(PSE, OrigLoop,
                           /*CanUseConstantMax=*/true,
-                          /*CanExcludeZeroTrips=*/CM->isEpilogueAllowed());
+                          /*CanExcludeZeroTrips=*/EnabledCM->isEpilogueAllowed());
 
   // For fixed length VFs treat a scalable trip count as unknown.
   if (BestKnownTC && (BestKnownTC->isFixed() || VF.isScalable())) {
@@ -5343,10 +5343,10 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
 }
 
 void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
-  CM->collectValuesToIgnore();
-  Config.collectElementTypesForWidening(&CM->ValuesToIgnore);
+  EnabledCM->collectValuesToIgnore();
+  Config.collectElementTypesForWidening(&EnabledCM->ValuesToIgnore);
 
-  FixedScalableVFPair MaxFactors = CM->computeMaxVF(UserVF, UserIC);
+  FixedScalableVFPair MaxFactors = EnabledCM->computeMaxVF(UserVF, UserIC);
   if (!MaxFactors) // Cases that should not to be vectorized nor interleaved.
     return;
 
@@ -5357,7 +5357,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   if (MaxFactors.FixedVF.isVector() || MaxFactors.ScalableVF.isVector())
     Legal->collectUnitStridePredicates();
 
-  auto VPlan1 = tryToBuildVPlan1(*CM);
+  auto VPlan1 = tryToBuildVPlan1();
   if (!VPlan1)
     return;
 
@@ -5366,7 +5366,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
     // plan for that VF only.
     ElementCount VF =
         MaxFactors.FixedVF ? MaxFactors.FixedVF : MaxFactors.ScalableVF;
-    buildVPlans(*VPlan1, VF, VF, *CM);
+    buildVPlans(*VPlan1, VF, VF);
     LLVM_DEBUG(printPlans(dbgs()));
     return;
   }
@@ -5376,20 +5376,20 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   Config.computeMinimalBitwidths();
 
   // Invalidate interleave groups if all blocks of loop will be predicated.
-  if (CM->blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
+  if (EnabledCM->blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
       !useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
         dbgs()
         << "LV: Invalidate all interleaved groups due to fold-tail by masking "
            "which requires masked-interleaved support.\n");
-    if (CM->InterleaveInfo.invalidateGroups())
+    if (EnabledCM->InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      CM->invalidateCostModelingDecisions();
+      EnabledCM->invalidateCostModelingDecisions();
   }
 
-  if (CM->foldTailByMasking())
+  if (EnabledCM->foldTailByMasking())
     Legal->prepareToFoldTailByMasking();
 
   ElementCount MaxUserVF =
@@ -5404,23 +5404,23 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
              "VF needs to be a power of two");
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
-      CM->collectNonVectorizedAndSetWideningDecisions(UserVF);
+      EnabledCM->collectNonVectorizedAndSetWideningDecisions(UserVF);
       // Build the main-loop VPlan firstly because if epilogue tail-folding is
       // enabled, it will be built later, so we keep the epilogue vplans at the
       // end.
-      buildVPlans(*VPlan1, UserVF, UserVF, *CM);
+      buildVPlans(*VPlan1, UserVF, UserVF);
 
       ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
-        CM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, *CM);
+        EnabledCM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
+        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF);
       }
       if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
         // vector VFs only.
         if (UserVF.isScalar() ||
-            cost(*VPlans.front(), UserVF, /*RU=*/nullptr, *CM).isValid()) {
+            cost(*VPlans.front(), UserVF, /*RU=*/nullptr).isValid()) {
           LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
           LLVM_DEBUG(printPlans(dbgs()));
           return;
@@ -5443,56 +5443,52 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 
   for (const auto &VF : VFCandidates) {
     // Collect Uniform and Scalar instructions after vectorization with VF.
-    CM->collectNonVectorizedAndSetWideningDecisions(VF);
+    EnabledCM->collectNonVectorizedAndSetWideningDecisions(VF);
   }
 
-  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF, *CM);
-  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
-              *CM);
+  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF);
+  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF);
 
   LLVM_DEBUG(printPlans(dbgs()));
 }
 
-bool LoopVectorizationPlanner::planForEpilogueTF(
-    ElementCount UserVF, unsigned UserIC,
-    LoopVectorizationCostModel &EpilogueCM) {
+bool LoopVectorizationPlanner::planForEpilogueTF(ElementCount UserVF,
+                                                 unsigned UserIC) {
   if (VPlans.empty()) {
     LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
                          "out of epilogue tail-folding\n");
     return false;
   }
 
-  EpilogueCM.ValuesToIgnore.insert_range(CM->ValuesToIgnore);
-  EpilogueCM.VecValuesToIgnore.insert_range(CM->VecValuesToIgnore);
+  EnabledCM->ValuesToIgnore.insert_range(CM->ValuesToIgnore);
+  EnabledCM->VecValuesToIgnore.insert_range(CM->VecValuesToIgnore);
 
   FixedScalableVFPair MaxFactors =
-      EpilogueCM.computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
-  if (!MaxFactors ||
-      !EpilogueCM
-           .foldTailByMasking()) { // Cases that should not to be vectorized
-                                   // or tail-folded.
+      EnabledCM->computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
+  if (!MaxFactors || !EnabledCM->foldTailByMasking()) {
+    // Cases that should not to be vectorized // or tail-folded.
     reportVectorizationInfo("This case of epilogue loop can't be tail-folded",
                             "InvalidTailFoldedEpilogue", ORE, OrigLoop);
     return false;
   }
 
-  auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
+  auto VPlan1 = tryToBuildVPlan1();
 
   if (!useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
         dbgs() << "LV: Invalidate all interleaved groups due to fold-tail by "
                   "masking which requires masked-interleaved support.\n");
-    if (EpilogueCM.InterleaveInfo.invalidateGroups())
+    if (EnabledCM->InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      EpilogueCM.invalidateCostModelingDecisions();
+      EnabledCM->invalidateCostModelingDecisions();
   }
   Legal->prepareToFoldTailByMasking();
 
   // Collect the instructions (and their associated costs) that will be more
   // profitable to scalarize.
-  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(
+  EnabledCM->collectNonVectorizedAndSetWideningDecisions(
       EpilogueVectorizationForceVF);
 
   // Expecting only 2 vplans as epilogue TF is applied only for forced VFs.
@@ -5505,10 +5501,9 @@ bool LoopVectorizationPlanner::planForEpilogueTF(
          "EpilogueUserVF");
   VPlans.pop_back();
   buildVPlans(*VPlan1, EpilogueVectorizationForceVF,
-              EpilogueVectorizationForceVF, EpilogueCM);
+              EpilogueVectorizationForceVF);
 
-  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr,
-       EpilogueCM);
+  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr);
   return true;
 }
 
@@ -5655,10 +5650,9 @@ getRecordedExecutionFrequency(const VPBasicBlock *VPBB) {
 }
 #endif
 
-InstructionCost LoopVectorizationPlanner::cost(
-    VPlan &Plan, ElementCount VF, VPRegisterUsage *RU,
-    LoopVectorizationCostModel &EnabledCM) const {
-  VPCostContext CostCtx(*TLI, Plan, EnabledCM, Config,
+InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
+                                               VPRegisterUsage *RU) const {
+  VPCostContext CostCtx(*TLI, Plan, *EnabledCM, Config,
                         /*ReusePrintingSlotTracker=*/true);
   InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
 
@@ -5733,7 +5727,7 @@ LoopVectorizationPlanner::computeBestVF() {
          "More than a single plan/VF w/o any plan having scalar VF");
 
   // TODO: Compute scalar cost using VPlan-based cost model.
-  InstructionCost ScalarCost = CM->expectedCost(ScalarVF);
+  InstructionCost ScalarCost = EnabledCM->expectedCost(ScalarVF);
   LLVM_DEBUG(dbgs() << "LV: Scalar loop costs: " << ScalarCost << ".\n");
   VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
   VectorizationFactor BestFactor = ScalarFactor;
@@ -5792,7 +5786,7 @@ LoopVectorizationPlanner::computeBestVF() {
       }
 
       InstructionCost Cost =
-          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr, *CM);
+          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
       if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
@@ -5818,12 +5812,14 @@ LoopVectorizationPlanner::computeBestVF() {
 LoopVectorizationPlanner::LoopVectorizationPlanner(
     Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
     const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
-    std::unique_ptr<LoopVectorizationCostModel> CM, VFSelectionContext &Config,
+    std::unique_ptr<LoopVectorizationCostModel> CM,
+    LoopVectorizationCostModel *EpilogueTFCM, VFSelectionContext &Config,
     InterleavedAccessInfo &IAI, PredicatedScalarEvolution &PSE,
     OptimizationRemarkEmitter *ORE,
     std::function<const BranchProbabilityInfo &()> GetBPI)
     : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal),
-      CM(std::move(CM)), Config(Config), IAI(IAI), PSE(PSE), ORE(ORE),
+      CM(std::move(CM)), EpilogueTFCM(EpilogueTFCM), Config(Config), IAI(IAI),
+      PSE(PSE), ORE(ORE),
       GetBPI(GetBPI) {}
 
 LoopVectorizationPlanner::~LoopVectorizationPlanner() = default;
@@ -6477,7 +6473,7 @@ static bool verifyExecutionFrequenciesMatchBFI(VPlan &Plan, Loop *OrigLoop,
 }
 #endif
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(LoopVectorizationCostModel &EnabledCM) {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
   bool IsInnerLoop = OrigLoop->isInnermost();
 
   // Set up loop versioning for inner loops with memory runtime checks.
@@ -6533,8 +6529,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(LoopVectorizationCostModel &
       Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled;
   bool OptForSize =
       !ForceVectorization &&
-      (EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
-       EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
+      (EnabledCM->EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
+       EnabledCM->EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
   unsigned SCEVCheckThreshold = ForceVectorization
                                     ? PragmaVectorizeSCEVCheckThreshold
                                     : VectorizeSCEVCheckThreshold;
@@ -6566,7 +6562,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(LoopVectorizationCostModel &
 
   RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
                  getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
-  if (EnabledCM.foldTailByMasking())
+  if (EnabledCM->foldTailByMasking())
     RUN_VPLAN_PASS(VPlanTransforms::foldTailByMasking, *VPlan0);
 
   RUN_VPLAN_PASS(VPlanTransforms::introduceMasksAndLinearize, *VPlan0);
@@ -6574,17 +6570,16 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(LoopVectorizationCostModel &
   return VPlan0;
 }
 
-void LoopVectorizationPlanner::buildVPlans(
-    VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
-    LoopVectorizationCostModel &EnabledCM) {
+void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
+                                           ElementCount MaxVF) {
   if (ElementCount::isKnownGT(MinVF, MaxVF))
     return;
 
   auto MaxVFTimes2 = MaxVF * 2;
   for (ElementCount VF = MinVF; ElementCount::isKnownLT(VF, MaxVFTimes2);) {
     VFRange SubRange = {VF, MaxVFTimes2};
-    auto Plan = tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()),
-                                SubRange, EnabledCM);
+    auto Plan =
+        tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()), SubRange);
     VF = SubRange.End;
 
     if (!Plan)
@@ -6597,7 +6592,7 @@ void LoopVectorizationPlanner::buildVPlans(
                    Config.getMinimalBitwidths());
     RUN_VPLAN_PASS(VPlanTransforms::optimize, *Plan);
     // TODO: try to put addExplicitVectorLength close to addActiveLaneMask
-    if (EnabledCM.foldTailWithEVL()) {
+    if (EnabledCM->foldTailWithEVL()) {
       RUN_VPLAN_PASS(VPlanTransforms::addExplicitVectorLength, *Plan,
                      Config.getMaxSafeElements());
       RUN_VPLAN_PASS(VPlanTransforms::optimizeEVLMasks, *Plan);
@@ -6607,7 +6602,7 @@ void LoopVectorizationPlanner::buildVPlans(
             RUN_VPLAN_PASS(VPlanTransforms::narrowInterleaveGroups, *Plan, TTI))
       VPlans.push_back(std::move(P));
 
-    TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
+    TailFoldingStyle Style = EnabledCM->getTailFoldingStyle();
     RUN_VPLAN_PASS(VPlanTransforms::materializeHeaderMask, *Plan,
                    useActiveLaneMask(Style),
                    useActiveLaneMaskForControlFlow(Style));
@@ -6618,8 +6613,8 @@ void LoopVectorizationPlanner::buildVPlans(
   }
 }
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
-    VPlanPtr Plan, VFRange &Range, LoopVectorizationCostModel &EnabledCM) {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
+                                                   VFRange &Range) {
 
   // For outer loops, the plan only needs basic recipe conversion and induction
   // live-out optimization; the full inner-loop recipe building below does not
@@ -6645,8 +6640,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
 
   bool RequiresScalarEpilogueCheck =
       LoopVectorizationPlanner::getDecisionAndClampRange(
-          [&EnabledCM](ElementCount VF) {
-            return !EnabledCM.requiresScalarEpilogue(VF.isVector());
+          [&](ElementCount VF) {
+            return !EnabledCM->requiresScalarEpilogue(VF.isVector());
           },
           Range);
   // Update the branch in the middle block if a scalar epilogue is required.
@@ -6664,9 +6659,9 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
   // TODO: Consider using getDecisionAndClampRange here to split up VPlans.
   bool IVUpdateMayOverflow = false;
   for (ElementCount VF : Range)
-    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(&EnabledCM, VF);
+    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(EnabledCM, VF);
 
-  TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
+  TailFoldingStyle Style = EnabledCM->getTailFoldingStyle();
   // Use NUW for the induction increment if we proved that it won't overflow in
   // the vector loop or when not folding the tail. In the later case, we know
   // that the canonical induction increment will not overflow as the vector trip
@@ -6693,10 +6688,10 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
   // placeholders for its members' Recipes which we'll be replacing with a
   // single VPInterleaveRecipe.
   for (InterleaveGroup<Instruction> *IG :
-       EnabledCM.InterleaveInfo.getInterleaveGroups()) {
-    auto ApplyIG = [IG, &EnabledCM](ElementCount VF) -> bool {
+       EnabledCM->InterleaveInfo.getInterleaveGroups()) {
+    auto ApplyIG = [IG, this](ElementCount VF) -> bool {
       bool Result = (VF.isVector() && // Query is illegal for VF == 1
-                     EnabledCM.getWideningDecision(IG->getInsertPos(), VF) ==
+                     EnabledCM->getWideningDecision(IG->getInsertPos(), VF) ==
                          LoopVectorizationCostModel::CM_Interleave);
       // For scalable vectors, the interleave factors must be <= 8 since we
       // require the (de)interleaveN intrinsics instead of shufflevectors.
@@ -6713,12 +6708,12 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
   // Construct wide recipes and apply predication for original scalar
   // VPInstructions in the loop.
   // ---------------------------------------------------------------------------
-  VPRecipeBuilder RecipeBuilder(*Plan, Legal, EnabledCM, Builder);
+  VPRecipeBuilder RecipeBuilder(*Plan, Legal, *EnabledCM, Builder);
 
   RUN_VPLAN_PASS(VPlanTransforms::createInLoopReductionRecipes, *Plan,
                  Range.Start);
 
-  VPCostContext CostCtx(*TLI, *Plan, EnabledCM, Config);
+  VPCostContext CostCtx(*TLI, *Plan, *EnabledCM, Config);
 
   RUN_VPLAN_PASS(VPlanTransforms::makeMemOpWideningDecisions, *Plan, Range,
                  RecipeBuilder, CostCtx);
@@ -6825,7 +6820,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
   // for this VPlan, replace the Recipes widening its memory instructions with a
   // single VPInterleaveRecipe at its insertion point.
   RUN_VPLAN_PASS(VPlanTransforms::createInterleaveGroups, *Plan,
-                 InterleaveGroups, EnabledCM.isEpilogueAllowed());
+                 InterleaveGroups, EnabledCM->isEpilogueAllowed());
 
   // Convert memory recipes to strided access recipes if the strided access is
   // legal and profitable.
@@ -6842,7 +6837,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
 
   RUN_VPLAN_PASS(VPlanTransforms::dropPoisonGeneratingRecipes, *Plan);
 
-  if (EnabledCM.maskPartialAliasing())
+  if (EnabledCM->maskPartialAliasing())
     RUN_VPLAN_PASS(VPlanTransforms::attachAliasMaskToHeaderMask, *Plan);
 
   assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
@@ -6893,7 +6888,7 @@ void LoopVectorizationPlanner::addReductionResultComputation(
 
     // Remove the predicated select if the target doesn't want it.
     VPValue *V;
-    if (!CM->usePredicatedReductionSelect(RecurrenceKind) &&
+    if (!EnabledCM->usePredicatedReductionSelect(RecurrenceKind) &&
         match(PhiR->getBackedgeValue(),
               m_Select(m_Specific(HeaderMask), m_VPValue(V), m_Specific(PhiR))))
       PhiR->setBackedgeValue(V);
@@ -8029,15 +8024,11 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Use the cost model.
   VFSelectionContext Config(*TTI, &LVL, L, *F, PSE, DB, ORE, &Hints,
                             OptForSize);
-  // Use the planner for vectorization.
-  LoopVectorizationPlanner LVP(
-      L, LI, DT, TLI, *TTI, &LVL,
-      std::make_unique<LoopVectorizationCostModel>(
-          SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config),
-      Config, IAI, PSE, ORE, GetBPI);
+  std::unique_ptr<LoopVectorizationCostModel> CM = std::make_unique<LoopVectorizationCostModel>(
+          SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config);
 
   EpilogueLowering EpilogueTailLoweringStatus =
-      getEpilogueTailLowering(LVP.getCostModel(), L, ORE, LVL, Hints, TTI);
+      getEpilogueTailLowering(*CM, L, ORE, LVL, Hints, TTI);
   std::optional<InterleavedAccessInfo> TailFoldingCMIAI;
   std::optional<LoopVectorizationCostModel> EpilogueTailFoldingCM;
   if (EpilogueTailLoweringStatus ==
@@ -8050,6 +8041,11 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                                   &LVL, *TTI, TLI, AC, ORE, GetBFI, F,
                                   *TailFoldingCMIAI, Config);
   }
+  // Use the planner for vectorization.
+  LoopVectorizationPlanner LVP(
+      L, LI, DT, TLI, *TTI, &LVL, std::move(CM), EpilogueTailFoldingCM ? &*EpilogueTailFoldingCM
+                                                     : nullptr,
+      Config, IAI, PSE, ORE, GetBPI);
 
   // Get user vectorization factor and interleave count.
   ElementCount UserVF = Hints.getWidth();
@@ -8063,14 +8059,19 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC);
-  if (EpilogueTailFoldingCM)
-    if (!LVP.planForEpilogueTF(UserVF, UserIC, EpilogueTailFoldingCM.value())) {
+  if (EpilogueTailFoldingCM) {
+    // Enable the epilogue tail-folding CM
+    LVP.enableEpilogueTFCM();
+    if (!LVP.planForEpilogueTF(UserVF, UserIC)) {
       // we can't apply epilogue TF:
       reportVectorizationInfo(
           "Applying epilogue tail-folding failed, disable it.",
           "InvalidTailFoldedEpilogue", ORE, L);
       EpilogueTailFoldingCM.reset();
     }
+    // Get back the default CM:
+    LVP.enableDefaultCM();
+  }
 
   auto [VF, BestPlanPtr] = LVP.computeBestVF();
   unsigned IC = 1;
@@ -8083,11 +8084,11 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   if (IsInnerLoop && ORE->allowExtraAnalysis(LV_NAME))
     LVP.emitInvalidCostRemarks(ORE);
 
-  assert((IsInnerLoop || !LVP.getCostModel().maskPartialAliasing()) &&
+  assert((IsInnerLoop || !CM->maskPartialAliasing()) &&
          "Did not expect to alias-mask outer loop");
 
   GeneratedRTChecks Checks(PSE, DT, LI, TTI, Config.CostKind,
-                           LVP.getCostModel().maskPartialAliasing());
+                           CM->maskPartialAliasing());
   if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width)) {
     // Select the interleave count.
     IC = LVP.selectInterleaveCount(*BestPlanPtr, VF.Width, VF.Cost);
@@ -8113,7 +8114,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     // Check if it is profitable to vectorize with runtime checks.
     bool ForceVectorization =
         Hints.getForce() == LoopVectorizeHints::FK_Enabled;
-    VPCostContext CostCtx(*TLI, *BestPlanPtr, LVP.getCostModel(), Config,
+    VPCostContext CostCtx(*TLI, *BestPlanPtr, *CM, Config,
                           /*ReusePrintingSlotTracker=*/true);
     if (!ForceVectorization &&
         !isOutsideLoopWorkProfitable(Checks, VF, L, PSE, CostCtx, *BestPlanPtr,
@@ -8196,7 +8197,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Override IC if user provided an interleave count.
   IC = UserIC > 0 ? UserIC : IC;
 
-  if (LVP.getCostModel().maskPartialAliasing()) {
+  if (CM->maskPartialAliasing()) {
     LLVM_DEBUG(
         dbgs()
         << "LV: Not interleaving due to partial aliasing vectorization.\n");
@@ -8276,11 +8277,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Whether a scalar epilogue may be created is decided by the epilogue
   // lowering policy.
   // TODO: Also move check to be based on VPlan.
-  bool ScalarEpilogueAllowed = LVP.getCostModel().isEpilogueAllowed();
-
-  // Destroy the cost model before executing any plan, so that code generation
-  // cannot rely on cost-modeling decisions.
-  LVP.clearCostModel();
+  bool ScalarEpilogueAllowed = CM->isEpilogueAllowed();
 
   VPlan &BestPlan = *BestPlanPtr;
   // Consider vectorizing the epilogue too if it's profitable.
@@ -8320,6 +8317,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
         LoopVectorizationPlanner::EpilogueVectorizationKind::MainLoop);
     ++LoopsVectorized;
 
+    if (EpilogueTailFoldingCM)
+      LVP.enableEpilogueTFCM();
     BasicBlock *EntryBB =
         cast<VPIRBasicBlock>(BestMainPlan.getEntry())->getIRBasicBlock();
     EntryBB->setName("iter.check");
@@ -8340,6 +8339,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                               InstsToMove, ResumeValues,
                               EpilogueTailFoldingCM.has_value());
     ++LoopsEpilogueVectorized;
+    LVP.enableDefaultCM();
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
                            BestPlan);
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index e9b248016c32f..ece1fc3a15b11 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,5 +1,5 @@
-; REQUIRES: asserts
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; REQUIRES: asserts
 ; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug-only=loop-vectorize -mcpu=neoverse-v1 -S %s | FileCheck %s
 

>From 41dea38d5fc6c7c0ab3a17720c670bc45bd597b0 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Mon, 10 Aug 2026 21:03:58 +0100
Subject: [PATCH 07/25] resolve review comments - improve readability

---
 .../Vectorize/LoopVectorizationPlanner.h      | 19 +++++++++++++------
 .../Transforms/Vectorize/LoopVectorize.cpp    |  5 ++---
 2 files changed, 15 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 7d3fc5e52e076..f53e9e4c9cfea 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -885,9 +885,17 @@ class LoopVectorizationPlanner {
   /// The legality analysis.
   LoopVectorizationLegality *Legal;
 
-  /// The profitability analysis. Cleared after making cost based decisions.
-  std::unique_ptr<LoopVectorizationCostModel> CM;
+  /// The profitability analysis.
+  /// The CM currently in effect for the VPlan being built or costed; it always
+  /// aliases either \c CM or \c EpilogueTFCM.
   LoopVectorizationCostModel *EnabledCM;
+  /// The CM used for the main-loop VPlan, and for the epilogue VPlan in all
+  /// cases except tail-folded epilogue vectorization.
+  std::unique_ptr<LoopVectorizationCostModel> CM;
+  /// The CM used only when the epilogue loop is vectorized with tail-folding.
+  /// \c EnabledCM is switched to point here (via enableEpilogueTFCM()) while
+  /// the epilogue VPlan's costs are computed, so that they correctly account
+  /// for the tail-folded epilogue.
   LoopVectorizationCostModel *EpilogueTFCM;
   /// VF selection state independent of cost-modeling decisions.
   VFSelectionContext &Config;
@@ -959,10 +967,9 @@ class LoopVectorizationPlanner {
   /// interleaving should be avoided up-front, no plans are generated.
   void plan(ElementCount UserVF, unsigned UserIC);
 
-  /// Build VPlans for the specified \p EpilogueUserVF and \p IC if they are
-  /// non-zero or all applicable candidate VFs otherwise. If vectorization and
-  /// tail-folding should be avoided up-front, no plans are generated.
-  bool planForEpilogueTF(ElementCount UserVF, unsigned UserIC);
+  /// Build VPlan for the forced epilogue VF. If vectorization and tail-folding
+  /// should be avoided up-front, no tail-folded plans are generated.
+  bool planForEpilogueTF();
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 9de61f2f09af8..8c3606f5c9fb3 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5452,8 +5452,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   LLVM_DEBUG(printPlans(dbgs()));
 }
 
-bool LoopVectorizationPlanner::planForEpilogueTF(ElementCount UserVF,
-                                                 unsigned UserIC) {
+bool LoopVectorizationPlanner::planForEpilogueTF() {
   if (VPlans.empty()) {
     LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
                          "out of epilogue tail-folding\n");
@@ -8062,7 +8061,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   if (EpilogueTailFoldingCM) {
     // Enable the epilogue tail-folding CM
     LVP.enableEpilogueTFCM();
-    if (!LVP.planForEpilogueTF(UserVF, UserIC)) {
+    if (!LVP.planForEpilogueTF()) {
       // we can't apply epilogue TF:
       reportVectorizationInfo(
           "Applying epilogue tail-folding failed, disable it.",

>From 6af87ebad2f81c307b9d8dcee2f7fd240f787b9b Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Tue, 11 Aug 2026 15:59:31 +0100
Subject: [PATCH 08/25] disallow epilogue TF for outer loop

---
 .../LoopVectorize/fold-epilogue-tail.ll       | 59 ++++++++++---------
 1 file changed, 31 insertions(+), 28 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index c4a7097411e3c..cff4c925366ab 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -91,6 +91,7 @@ exit:
 define i16 @require_scalar_epilogue(ptr %dst, i64 %x) {
 ; CHECK-LABEL: Checking a loop in 'require_scalar_epilogue'
 ; CHECK: remark: <unknown>:0:0: Epilogue tail-folding can't be applied because scalar epilogue is required. Fall back to a normal epilogue
+; CHECK-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %loop.header
@@ -119,6 +120,7 @@ exit.2:
 define i32 @opt_for_size(ptr %p, i32 %n, i8 %val) optsize {
 ; CHECK-LABEL: Checking a loop in 'opt_for_size'
 ; CHECK: remark: <unknown>:0:0: Not applying tail-folding to the epilogue, since no epilogue is allowed
+; CHECK-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:
   br label %for.body
@@ -277,34 +279,6 @@ for.end:
   ret void
 }
 
-define void @test_outer_loop(ptr %A, i64 %m) {
-; CHECK-OUTER-LOOP-LABEL: Checking a loop in 'test_outer_loop'
-; CHECK-OUTER-LOOP: remark: <unknown>:0:0: Epilogue tail-folding is not supported for outer loop
-;
-entry:
-  br label %outer.header
-
-outer.header:
-  %iv.outer = phi i64 [ 0, %entry ], [ %iv.outer.next, %outer.latch ]
-  br label %inner
-
-inner:
-  %iv.inner = phi i64 [ 0, %outer.header ], [ %iv.inner.next, %inner ]
-  %gep = getelementptr inbounds i8, ptr %A, i64 %iv.inner
-  store i8 0, ptr %gep, align 1
-  %iv.inner.next = add nuw nsw i64 %iv.inner, 1
-  %inner.ec = icmp eq i64 %iv.inner.next, 8
-  br i1 %inner.ec, label %outer.latch, label %inner
-
-outer.latch:
-  %iv.outer.next = add nuw nsw i64 %iv.outer, 1
-  %outer.ec = icmp eq i64 %iv.outer.next, %m
-  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !1
-
-exit:
-  ret void
-}
-
 ; Can't build a valid vplan for this case because too many SCEV checks needed,
 ; more than the specfied limit.
 define i64 @test_no_vplan_built(ptr %dst, i64 %n) {
@@ -332,5 +306,34 @@ exit:
   ret i64 %result
 }
 
+define void @test_outer_loop(ptr %A, i64 %m) {
+; CHECK-OUTER-LOOP-LABEL: Checking a loop in 'test_outer_loop'
+; CHECK-OUTER-LOOP: remark: <unknown>:0:0: Epilogue tail-folding is not supported for outer loop
+; CHECK-OUTER-LOOP-NOT: LV: epilogue tail-folding is enabled
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %iv.outer = phi i64 [ 0, %entry ], [ %iv.outer.next, %outer.latch ]
+  br label %inner
+
+inner:
+  %iv.inner = phi i64 [ 0, %outer.header ], [ %iv.inner.next, %inner ]
+  %gep = getelementptr inbounds i32, ptr %A, i64 %iv.inner
+  store i32 0, ptr %gep, align 4
+  %iv.inner.next = add nuw nsw i64 %iv.inner, 1
+  %inner.ec = icmp eq i64 %iv.inner.next, 8
+  br i1 %inner.ec, label %outer.latch, label %inner
+
+outer.latch:
+  %iv.outer.next = add nuw nsw i64 %iv.outer, 1
+  %outer.ec = icmp eq i64 %iv.outer.next, %m
+  br i1 %outer.ec, label %exit, label %outer.header, !llvm.loop !1
+
+exit:
+  ret void
+}
+
 !1 = distinct !{!1, !2}
 !2 = !{!"llvm.loop.vectorize.enable"}

>From 90d48220e79fa87d0bef4268f871f36fe4890b39 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Fri, 14 Aug 2026 16:57:33 +0100
Subject: [PATCH 09/25] revert commit of swapping between different CMs

---
 .../Vectorize/LoopVectorizationPlanner.h      |  34 +---
 .../Transforms/Vectorize/LoopVectorize.cpp    | 180 +++++++++---------
 2 files changed, 100 insertions(+), 114 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index f53e9e4c9cfea..739c416ed3a5a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -885,18 +885,9 @@ class LoopVectorizationPlanner {
   /// The legality analysis.
   LoopVectorizationLegality *Legal;
 
-  /// The profitability analysis.
-  /// The CM currently in effect for the VPlan being built or costed; it always
-  /// aliases either \c CM or \c EpilogueTFCM.
-  LoopVectorizationCostModel *EnabledCM;
-  /// The CM used for the main-loop VPlan, and for the epilogue VPlan in all
-  /// cases except tail-folded epilogue vectorization.
+  /// The profitability analysis. Cleared after making cost based decisions.
   std::unique_ptr<LoopVectorizationCostModel> CM;
-  /// The CM used only when the epilogue loop is vectorized with tail-folding.
-  /// \c EnabledCM is switched to point here (via enableEpilogueTFCM()) while
-  /// the epilogue VPlan's costs are computed, so that they correctly account
-  /// for the tail-folded epilogue.
-  LoopVectorizationCostModel *EpilogueTFCM;
+
   /// VF selection state independent of cost-modeling decisions.
   VFSelectionContext &Config;
 
@@ -926,7 +917,8 @@ class LoopVectorizationPlanner {
   ///
   /// TODO: Move to VPlan::cost once the use of LoopVectorizationLegality has
   /// been retired.
-  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU) const;
+  InstructionCost cost(VPlan &Plan, ElementCount VF, VPRegisterUsage *RU,
+                       LoopVectorizationCostModel &EnabledCM) const;
 
   /// Precompute costs for certain instructions using the legacy cost model. The
   /// function is used to bring up the VPlan-based cost model to initially avoid
@@ -939,7 +931,6 @@ class LoopVectorizationPlanner {
       Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
       const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
       std::unique_ptr<LoopVectorizationCostModel> CM,
-      LoopVectorizationCostModel *EpilogueTFCM,
       VFSelectionContext &Config, InterleavedAccessInfo &IAI,
       PredicatedScalarEvolution &PSE, OptimizationRemarkEmitter *ORE,
       std::function<const BranchProbabilityInfo &()> GetBPI);
@@ -955,13 +946,6 @@ class LoopVectorizationPlanner {
   /// Destroy the cost model.
   void clearCostModel();
 
-  void enableDefaultCM() { EnabledCM = &*CM; }
-
-  void enableEpilogueTFCM() {
-    assert(EpilogueTFCM && "No CM for epilogue tail-folding to enable");
-    EnabledCM = EpilogueTFCM;
-  }
-
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// interleaving should be avoided up-front, no plans are generated.
@@ -969,7 +953,7 @@ class LoopVectorizationPlanner {
 
   /// Build VPlan for the forced epilogue VF. If vectorization and tail-folding
   /// should be avoided up-front, no tail-folded plans are generated.
-  bool planForEpilogueTF();
+  bool planForEpilogueTF(LoopVectorizationCostModel &EpilogueCM);
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
@@ -1067,7 +1051,7 @@ class LoopVectorizationPlanner {
   /// Build an initial VPlan, with HCFG wrapping the original scalar loop and
   /// scalar transformations applied. Returns null if an initial VPlan cannot
   /// be built.
-  VPlanPtr tryToBuildVPlan1();
+  VPlanPtr tryToBuildVPlan1(LoopVectorizationCostModel &EnabledCM);
 
   /// Build a VPlan using VPRecipes according to the information gathered by
   /// Legal and VPlan-based analysis. For outer loops, performs basic recipe
@@ -1077,12 +1061,14 @@ class LoopVectorizationPlanner {
   /// maximum VF for which no plan could be built. Each VPlan is built starting
   /// from a copy of \p InitialPlan, which is a plain CFG VPlan wrapping the
   /// original scalar loop.
-  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range);
+  VPlanPtr tryToBuildVPlan(VPlanPtr InitialPlan, VFRange &Range,
+                           LoopVectorizationCostModel &EnabledCM);
 
   /// Build VPlans for power-of-2 VF's between \p MinVF and \p MaxVF inclusive,
   /// based on \p VPlan1 and according to the information gathered by Legal
   /// when it checked if it is legal to vectorize the loop.
-  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF);
+  void buildVPlans(VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
+                   LoopVectorizationCostModel &EnabledCM);
 
   /// Add ComputeReductionResult recipes to the middle block to compute the
   /// final reduction results. Add Select recipes to the latch block when
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 8c3606f5c9fb3..de3cc512922c9 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3172,7 +3172,7 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
       if (VF.isScalar())
         continue;
 
-      VPCostContext CostCtx(*TLI, *Plan, *EnabledCM, Config,
+      VPCostContext CostCtx(*TLI, *Plan, *CM, Config,
                             /*ReusePrintingSlotTracker=*/true);
       precomputeCosts(*Plan, VF, CostCtx);
       auto Iter = vp_depth_first_deep(Plan->getVectorLoopRegion()->getEntry());
@@ -3713,7 +3713,7 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   // Do not interleave tail-folded loops, as the overhead of multiple
   // instructions to calculate the predicate is likely not beneficial.
   // If an epilogue is not allowed for any other reason, do not interleave.
-  if (!EnabledCM->isEpilogueAllowed())
+  if (!CM->isEpilogueAllowed())
     return 1;
 
   if (any_of(Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis(),
@@ -3747,9 +3747,9 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   // then we calculate the cost of VF here.
   if (LoopCost == 0) {
     if (VF.isScalar())
-      LoopCost = EnabledCM->expectedCost(VF);
+      LoopCost = CM->expectedCost(VF);
     else
-      LoopCost = cost(Plan, VF, &R);
+      LoopCost = cost(Plan, VF, &R, *CM);
     assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
 
     // Loop body is free and there is no need for interleaving.
@@ -3830,7 +3830,7 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   auto BestKnownTC =
       getSmallBestKnownTC(PSE, OrigLoop,
                           /*CanUseConstantMax=*/true,
-                          /*CanExcludeZeroTrips=*/EnabledCM->isEpilogueAllowed());
+                          /*CanExcludeZeroTrips=*/CM->isEpilogueAllowed());
 
   // For fixed length VFs treat a scalable trip count as unknown.
   if (BestKnownTC && (BestKnownTC->isFixed() || VF.isScalable())) {
@@ -5343,10 +5343,10 @@ void LoopVectorizationCostModel::collectValuesToIgnore() {
 }
 
 void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
-  EnabledCM->collectValuesToIgnore();
-  Config.collectElementTypesForWidening(&EnabledCM->ValuesToIgnore);
+  CM->collectValuesToIgnore();
+  Config.collectElementTypesForWidening(&CM->ValuesToIgnore);
 
-  FixedScalableVFPair MaxFactors = EnabledCM->computeMaxVF(UserVF, UserIC);
+  FixedScalableVFPair MaxFactors = CM->computeMaxVF(UserVF, UserIC);
   if (!MaxFactors) // Cases that should not to be vectorized nor interleaved.
     return;
 
@@ -5357,7 +5357,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   if (MaxFactors.FixedVF.isVector() || MaxFactors.ScalableVF.isVector())
     Legal->collectUnitStridePredicates();
 
-  auto VPlan1 = tryToBuildVPlan1();
+  auto VPlan1 = tryToBuildVPlan1(*CM);
   if (!VPlan1)
     return;
 
@@ -5366,7 +5366,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
     // plan for that VF only.
     ElementCount VF =
         MaxFactors.FixedVF ? MaxFactors.FixedVF : MaxFactors.ScalableVF;
-    buildVPlans(*VPlan1, VF, VF);
+    buildVPlans(*VPlan1, VF, VF, *CM);
     LLVM_DEBUG(printPlans(dbgs()));
     return;
   }
@@ -5376,20 +5376,20 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   Config.computeMinimalBitwidths();
 
   // Invalidate interleave groups if all blocks of loop will be predicated.
-  if (EnabledCM->blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
+  if (CM->blockNeedsPredicationForAnyReason(OrigLoop->getHeader()) &&
       !useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
         dbgs()
         << "LV: Invalidate all interleaved groups due to fold-tail by masking "
            "which requires masked-interleaved support.\n");
-    if (EnabledCM->InterleaveInfo.invalidateGroups())
+    if (CM->InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      EnabledCM->invalidateCostModelingDecisions();
+      CM->invalidateCostModelingDecisions();
   }
 
-  if (EnabledCM->foldTailByMasking())
+  if (CM->foldTailByMasking())
     Legal->prepareToFoldTailByMasking();
 
   ElementCount MaxUserVF =
@@ -5404,23 +5404,23 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
              "VF needs to be a power of two");
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
-      EnabledCM->collectNonVectorizedAndSetWideningDecisions(UserVF);
+      CM->collectNonVectorizedAndSetWideningDecisions(UserVF);
       // Build the main-loop VPlan firstly because if epilogue tail-folding is
       // enabled, it will be built later, so we keep the epilogue vplans at the
       // end.
-      buildVPlans(*VPlan1, UserVF, UserVF);
+      buildVPlans(*VPlan1, UserVF, UserVF, *CM);
 
       ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
-        EnabledCM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF);
+        CM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
+        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, *CM);
       }
       if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
         // For scalar VF, skip VPlan cost check as VPlan cost is designed for
         // vector VFs only.
         if (UserVF.isScalar() ||
-            cost(*VPlans.front(), UserVF, /*RU=*/nullptr).isValid()) {
+            cost(*VPlans.front(), UserVF, /*RU=*/nullptr, *CM).isValid()) {
           LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
           LLVM_DEBUG(printPlans(dbgs()));
           return;
@@ -5443,51 +5443,55 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 
   for (const auto &VF : VFCandidates) {
     // Collect Uniform and Scalar instructions after vectorization with VF.
-    EnabledCM->collectNonVectorizedAndSetWideningDecisions(VF);
+    CM->collectNonVectorizedAndSetWideningDecisions(VF);
   }
 
-  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF);
-  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF);
+  buildVPlans(*VPlan1, ElementCount::getFixed(1), MaxFactors.FixedVF, *CM);
+  buildVPlans(*VPlan1, ElementCount::getScalable(1), MaxFactors.ScalableVF,
+              *CM);
 
   LLVM_DEBUG(printPlans(dbgs()));
 }
 
-bool LoopVectorizationPlanner::planForEpilogueTF() {
+bool LoopVectorizationPlanner::planForEpilogueTF(
+    LoopVectorizationCostModel &EpilogueCM) {
   if (VPlans.empty()) {
     LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
                          "out of epilogue tail-folding\n");
     return false;
   }
 
-  EnabledCM->ValuesToIgnore.insert_range(CM->ValuesToIgnore);
-  EnabledCM->VecValuesToIgnore.insert_range(CM->VecValuesToIgnore);
+  EpilogueCM.ValuesToIgnore.insert_range(CM->ValuesToIgnore);
+  EpilogueCM.VecValuesToIgnore.insert_range(CM->VecValuesToIgnore);
 
   FixedScalableVFPair MaxFactors =
-      EnabledCM->computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
-  if (!MaxFactors || !EnabledCM->foldTailByMasking()) {
-    // Cases that should not to be vectorized // or tail-folded.
+      EpilogueCM.computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
+  if (!MaxFactors ||
+      !EpilogueCM
+           .foldTailByMasking()) { // Cases that should not to be vectorized
+                                   // or tail-folded.
     reportVectorizationInfo("This case of epilogue loop can't be tail-folded",
                             "InvalidTailFoldedEpilogue", ORE, OrigLoop);
     return false;
   }
 
-  auto VPlan1 = tryToBuildVPlan1();
+  auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
 
   if (!useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
         dbgs() << "LV: Invalidate all interleaved groups due to fold-tail by "
                   "masking which requires masked-interleaved support.\n");
-    if (EnabledCM->InterleaveInfo.invalidateGroups())
+    if (EpilogueCM.InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      EnabledCM->invalidateCostModelingDecisions();
+      EpilogueCM.invalidateCostModelingDecisions();
   }
   Legal->prepareToFoldTailByMasking();
 
   // Collect the instructions (and their associated costs) that will be more
   // profitable to scalarize.
-  EnabledCM->collectNonVectorizedAndSetWideningDecisions(
+  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(
       EpilogueVectorizationForceVF);
 
   // Expecting only 2 vplans as epilogue TF is applied only for forced VFs.
@@ -5500,9 +5504,10 @@ bool LoopVectorizationPlanner::planForEpilogueTF() {
          "EpilogueUserVF");
   VPlans.pop_back();
   buildVPlans(*VPlan1, EpilogueVectorizationForceVF,
-              EpilogueVectorizationForceVF);
+              EpilogueVectorizationForceVF, EpilogueCM);
 
-  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr);
+  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr,
+       EpilogueCM);
   return true;
 }
 
@@ -5649,9 +5654,10 @@ getRecordedExecutionFrequency(const VPBasicBlock *VPBB) {
 }
 #endif
 
-InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
-                                               VPRegisterUsage *RU) const {
-  VPCostContext CostCtx(*TLI, Plan, *EnabledCM, Config,
+InstructionCost LoopVectorizationPlanner::cost(
+    VPlan &Plan, ElementCount VF, VPRegisterUsage *RU,
+    LoopVectorizationCostModel &EnabledCM) const {
+  VPCostContext CostCtx(*TLI, Plan, EnabledCM, Config,
                         /*ReusePrintingSlotTracker=*/true);
   InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
 
@@ -5726,7 +5732,7 @@ LoopVectorizationPlanner::computeBestVF() {
          "More than a single plan/VF w/o any plan having scalar VF");
 
   // TODO: Compute scalar cost using VPlan-based cost model.
-  InstructionCost ScalarCost = EnabledCM->expectedCost(ScalarVF);
+  InstructionCost ScalarCost = CM->expectedCost(ScalarVF);
   LLVM_DEBUG(dbgs() << "LV: Scalar loop costs: " << ScalarCost << ".\n");
   VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
   VectorizationFactor BestFactor = ScalarFactor;
@@ -5785,7 +5791,7 @@ LoopVectorizationPlanner::computeBestVF() {
       }
 
       InstructionCost Cost =
-          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
+          cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr, *CM);
       VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
 
       if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
@@ -5811,14 +5817,12 @@ LoopVectorizationPlanner::computeBestVF() {
 LoopVectorizationPlanner::LoopVectorizationPlanner(
     Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
     const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
-    std::unique_ptr<LoopVectorizationCostModel> CM,
-    LoopVectorizationCostModel *EpilogueTFCM, VFSelectionContext &Config,
+    std::unique_ptr<LoopVectorizationCostModel> CM, VFSelectionContext &Config,
     InterleavedAccessInfo &IAI, PredicatedScalarEvolution &PSE,
     OptimizationRemarkEmitter *ORE,
     std::function<const BranchProbabilityInfo &()> GetBPI)
     : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal),
-      CM(std::move(CM)), EpilogueTFCM(EpilogueTFCM), Config(Config), IAI(IAI),
-      PSE(PSE), ORE(ORE),
+      CM(std::move(CM)), Config(Config), IAI(IAI), PSE(PSE), ORE(ORE),
       GetBPI(GetBPI) {}
 
 LoopVectorizationPlanner::~LoopVectorizationPlanner() = default;
@@ -6472,7 +6476,7 @@ static bool verifyExecutionFrequenciesMatchBFI(VPlan &Plan, Loop *OrigLoop,
 }
 #endif
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(LoopVectorizationCostModel &EnabledCM) {
   bool IsInnerLoop = OrigLoop->isInnermost();
 
   // Set up loop versioning for inner loops with memory runtime checks.
@@ -6528,8 +6532,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
       Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled;
   bool OptForSize =
       !ForceVectorization &&
-      (EnabledCM->EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
-       EnabledCM->EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
+      (EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
+       EnabledCM.EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
   unsigned SCEVCheckThreshold = ForceVectorization
                                     ? PragmaVectorizeSCEVCheckThreshold
                                     : VectorizeSCEVCheckThreshold;
@@ -6561,7 +6565,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
 
   RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
                  getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
-  if (EnabledCM->foldTailByMasking())
+  if (EnabledCM.foldTailByMasking())
     RUN_VPLAN_PASS(VPlanTransforms::foldTailByMasking, *VPlan0);
 
   RUN_VPLAN_PASS(VPlanTransforms::introduceMasksAndLinearize, *VPlan0);
@@ -6569,16 +6573,17 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
   return VPlan0;
 }
 
-void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
-                                           ElementCount MaxVF) {
+void LoopVectorizationPlanner::buildVPlans(
+    VPlan &VPlan1, ElementCount MinVF, ElementCount MaxVF,
+    LoopVectorizationCostModel &EnabledCM) {
   if (ElementCount::isKnownGT(MinVF, MaxVF))
     return;
 
   auto MaxVFTimes2 = MaxVF * 2;
   for (ElementCount VF = MinVF; ElementCount::isKnownLT(VF, MaxVFTimes2);) {
     VFRange SubRange = {VF, MaxVFTimes2};
-    auto Plan =
-        tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()), SubRange);
+    auto Plan = tryToBuildVPlan(std::unique_ptr<VPlan>(VPlan1.duplicate()),
+                                SubRange, EnabledCM);
     VF = SubRange.End;
 
     if (!Plan)
@@ -6591,7 +6596,7 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
                    Config.getMinimalBitwidths());
     RUN_VPLAN_PASS(VPlanTransforms::optimize, *Plan);
     // TODO: try to put addExplicitVectorLength close to addActiveLaneMask
-    if (EnabledCM->foldTailWithEVL()) {
+    if (EnabledCM.foldTailWithEVL()) {
       RUN_VPLAN_PASS(VPlanTransforms::addExplicitVectorLength, *Plan,
                      Config.getMaxSafeElements());
       RUN_VPLAN_PASS(VPlanTransforms::optimizeEVLMasks, *Plan);
@@ -6601,7 +6606,7 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
             RUN_VPLAN_PASS(VPlanTransforms::narrowInterleaveGroups, *Plan, TTI))
       VPlans.push_back(std::move(P));
 
-    TailFoldingStyle Style = EnabledCM->getTailFoldingStyle();
+    TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
     RUN_VPLAN_PASS(VPlanTransforms::materializeHeaderMask, *Plan,
                    useActiveLaneMask(Style),
                    useActiveLaneMaskForControlFlow(Style));
@@ -6612,8 +6617,8 @@ void LoopVectorizationPlanner::buildVPlans(VPlan &VPlan1, ElementCount MinVF,
   }
 }
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
-                                                   VFRange &Range) {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(
+    VPlanPtr Plan, VFRange &Range, LoopVectorizationCostModel &EnabledCM) {
 
   // For outer loops, the plan only needs basic recipe conversion and induction
   // live-out optimization; the full inner-loop recipe building below does not
@@ -6639,8 +6644,8 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
 
   bool RequiresScalarEpilogueCheck =
       LoopVectorizationPlanner::getDecisionAndClampRange(
-          [&](ElementCount VF) {
-            return !EnabledCM->requiresScalarEpilogue(VF.isVector());
+          [&EnabledCM](ElementCount VF) {
+            return !EnabledCM.requiresScalarEpilogue(VF.isVector());
           },
           Range);
   // Update the branch in the middle block if a scalar epilogue is required.
@@ -6658,9 +6663,9 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // TODO: Consider using getDecisionAndClampRange here to split up VPlans.
   bool IVUpdateMayOverflow = false;
   for (ElementCount VF : Range)
-    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(EnabledCM, VF);
+    IVUpdateMayOverflow |= !isIndvarOverflowCheckKnownFalse(&EnabledCM, VF);
 
-  TailFoldingStyle Style = EnabledCM->getTailFoldingStyle();
+  TailFoldingStyle Style = EnabledCM.getTailFoldingStyle();
   // Use NUW for the induction increment if we proved that it won't overflow in
   // the vector loop or when not folding the tail. In the later case, we know
   // that the canonical induction increment will not overflow as the vector trip
@@ -6687,10 +6692,10 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // placeholders for its members' Recipes which we'll be replacing with a
   // single VPInterleaveRecipe.
   for (InterleaveGroup<Instruction> *IG :
-       EnabledCM->InterleaveInfo.getInterleaveGroups()) {
-    auto ApplyIG = [IG, this](ElementCount VF) -> bool {
+       EnabledCM.InterleaveInfo.getInterleaveGroups()) {
+    auto ApplyIG = [IG, &EnabledCM](ElementCount VF) -> bool {
       bool Result = (VF.isVector() && // Query is illegal for VF == 1
-                     EnabledCM->getWideningDecision(IG->getInsertPos(), VF) ==
+                     EnabledCM.getWideningDecision(IG->getInsertPos(), VF) ==
                          LoopVectorizationCostModel::CM_Interleave);
       // For scalable vectors, the interleave factors must be <= 8 since we
       // require the (de)interleaveN intrinsics instead of shufflevectors.
@@ -6707,12 +6712,12 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // Construct wide recipes and apply predication for original scalar
   // VPInstructions in the loop.
   // ---------------------------------------------------------------------------
-  VPRecipeBuilder RecipeBuilder(*Plan, Legal, *EnabledCM, Builder);
+  VPRecipeBuilder RecipeBuilder(*Plan, Legal, EnabledCM, Builder);
 
   RUN_VPLAN_PASS(VPlanTransforms::createInLoopReductionRecipes, *Plan,
                  Range.Start);
 
-  VPCostContext CostCtx(*TLI, *Plan, *EnabledCM, Config);
+  VPCostContext CostCtx(*TLI, *Plan, EnabledCM, Config);
 
   RUN_VPLAN_PASS(VPlanTransforms::makeMemOpWideningDecisions, *Plan, Range,
                  RecipeBuilder, CostCtx);
@@ -6819,7 +6824,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
   // for this VPlan, replace the Recipes widening its memory instructions with a
   // single VPInterleaveRecipe at its insertion point.
   RUN_VPLAN_PASS(VPlanTransforms::createInterleaveGroups, *Plan,
-                 InterleaveGroups, EnabledCM->isEpilogueAllowed());
+                 InterleaveGroups, EnabledCM.isEpilogueAllowed());
 
   // Convert memory recipes to strided access recipes if the strided access is
   // legal and profitable.
@@ -6836,7 +6841,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan(VPlanPtr Plan,
 
   RUN_VPLAN_PASS(VPlanTransforms::dropPoisonGeneratingRecipes, *Plan);
 
-  if (EnabledCM->maskPartialAliasing())
+  if (EnabledCM.maskPartialAliasing())
     RUN_VPLAN_PASS(VPlanTransforms::attachAliasMaskToHeaderMask, *Plan);
 
   assert(verifyVPlanIsValid(*Plan) && "VPlan is invalid");
@@ -6887,7 +6892,7 @@ void LoopVectorizationPlanner::addReductionResultComputation(
 
     // Remove the predicated select if the target doesn't want it.
     VPValue *V;
-    if (!EnabledCM->usePredicatedReductionSelect(RecurrenceKind) &&
+    if (!CM->usePredicatedReductionSelect(RecurrenceKind) &&
         match(PhiR->getBackedgeValue(),
               m_Select(m_Specific(HeaderMask), m_VPValue(V), m_Specific(PhiR))))
       PhiR->setBackedgeValue(V);
@@ -8023,11 +8028,15 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Use the cost model.
   VFSelectionContext Config(*TTI, &LVL, L, *F, PSE, DB, ORE, &Hints,
                             OptForSize);
-  std::unique_ptr<LoopVectorizationCostModel> CM = std::make_unique<LoopVectorizationCostModel>(
-          SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config);
+  // Use the planner for vectorization.
+  LoopVectorizationPlanner LVP(
+      L, LI, DT, TLI, *TTI, &LVL,
+      std::make_unique<LoopVectorizationCostModel>(
+          SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config),
+      Config, IAI, PSE, ORE, GetBPI);
 
   EpilogueLowering EpilogueTailLoweringStatus =
-      getEpilogueTailLowering(*CM, L, ORE, LVL, Hints, TTI);
+      getEpilogueTailLowering(LVP.getCostModel(), L, ORE, LVL, Hints, TTI);
   std::optional<InterleavedAccessInfo> TailFoldingCMIAI;
   std::optional<LoopVectorizationCostModel> EpilogueTailFoldingCM;
   if (EpilogueTailLoweringStatus ==
@@ -8040,11 +8049,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                                   &LVL, *TTI, TLI, AC, ORE, GetBFI, F,
                                   *TailFoldingCMIAI, Config);
   }
-  // Use the planner for vectorization.
-  LoopVectorizationPlanner LVP(
-      L, LI, DT, TLI, *TTI, &LVL, std::move(CM), EpilogueTailFoldingCM ? &*EpilogueTailFoldingCM
-                                                     : nullptr,
-      Config, IAI, PSE, ORE, GetBPI);
 
   // Get user vectorization factor and interleave count.
   ElementCount UserVF = Hints.getWidth();
@@ -8058,19 +8062,14 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC);
-  if (EpilogueTailFoldingCM) {
-    // Enable the epilogue tail-folding CM
-    LVP.enableEpilogueTFCM();
-    if (!LVP.planForEpilogueTF()) {
+  if (EpilogueTailFoldingCM)
+    if (!LVP.planForEpilogueTF(EpilogueTailFoldingCM.value())) {
       // we can't apply epilogue TF:
       reportVectorizationInfo(
           "Applying epilogue tail-folding failed, disable it.",
           "InvalidTailFoldedEpilogue", ORE, L);
       EpilogueTailFoldingCM.reset();
     }
-    // Get back the default CM:
-    LVP.enableDefaultCM();
-  }
 
   auto [VF, BestPlanPtr] = LVP.computeBestVF();
   unsigned IC = 1;
@@ -8083,11 +8082,11 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   if (IsInnerLoop && ORE->allowExtraAnalysis(LV_NAME))
     LVP.emitInvalidCostRemarks(ORE);
 
-  assert((IsInnerLoop || !CM->maskPartialAliasing()) &&
+  assert((IsInnerLoop || !LVP.getCostModel().maskPartialAliasing()) &&
          "Did not expect to alias-mask outer loop");
 
   GeneratedRTChecks Checks(PSE, DT, LI, TTI, Config.CostKind,
-                           CM->maskPartialAliasing());
+                           LVP.getCostModel().maskPartialAliasing());
   if (IsInnerLoop && LVP.hasPlanWithVF(VF.Width)) {
     // Select the interleave count.
     IC = LVP.selectInterleaveCount(*BestPlanPtr, VF.Width, VF.Cost);
@@ -8113,7 +8112,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     // Check if it is profitable to vectorize with runtime checks.
     bool ForceVectorization =
         Hints.getForce() == LoopVectorizeHints::FK_Enabled;
-    VPCostContext CostCtx(*TLI, *BestPlanPtr, *CM, Config,
+    VPCostContext CostCtx(*TLI, *BestPlanPtr, LVP.getCostModel(), Config,
                           /*ReusePrintingSlotTracker=*/true);
     if (!ForceVectorization &&
         !isOutsideLoopWorkProfitable(Checks, VF, L, PSE, CostCtx, *BestPlanPtr,
@@ -8196,7 +8195,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Override IC if user provided an interleave count.
   IC = UserIC > 0 ? UserIC : IC;
 
-  if (CM->maskPartialAliasing()) {
+  if (LVP.getCostModel().maskPartialAliasing()) {
     LLVM_DEBUG(
         dbgs()
         << "LV: Not interleaving due to partial aliasing vectorization.\n");
@@ -8276,7 +8275,11 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Whether a scalar epilogue may be created is decided by the epilogue
   // lowering policy.
   // TODO: Also move check to be based on VPlan.
-  bool ScalarEpilogueAllowed = CM->isEpilogueAllowed();
+  bool ScalarEpilogueAllowed = LVP.getCostModel().isEpilogueAllowed();
+
+  // Destroy the cost model before executing any plan, so that code generation
+  // cannot rely on cost-modeling decisions.
+  LVP.clearCostModel();
 
   VPlan &BestPlan = *BestPlanPtr;
   // Consider vectorizing the epilogue too if it's profitable.
@@ -8316,8 +8319,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
         LoopVectorizationPlanner::EpilogueVectorizationKind::MainLoop);
     ++LoopsVectorized;
 
-    if (EpilogueTailFoldingCM)
-      LVP.enableEpilogueTFCM();
     BasicBlock *EntryBB =
         cast<VPIRBasicBlock>(BestMainPlan.getEntry())->getIRBasicBlock();
     EntryBB->setName("iter.check");
@@ -8338,7 +8339,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                               InstsToMove, ResumeValues,
                               EpilogueTailFoldingCM.has_value());
     ++LoopsEpilogueVectorized;
-    LVP.enableDefaultCM();
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
                            BestPlan);

>From 36b224374923fa12c291ad5d9eacda84cba416e6 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Wed, 19 Aug 2026 16:06:42 +0100
Subject: [PATCH 10/25] Add test cases for different scenarios that should be
 supported by the feature

---
 .../AArch64/fold-epilogue-tail.ll             | 412 +++++++++++++++++-
 1 file changed, 399 insertions(+), 13 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index ece1fc3a15b11..79495c4ad9c3e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -8,9 +8,9 @@
 
 target triple = "aarch64-linux-gnu"
 
-define void @test_epilogue_tf(ptr %A, i64 %n) {
+define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-LABEL: define void @test_epilogue_tf(
-; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ITER_CHECK:.*]]:
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
@@ -20,13 +20,15 @@ define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <16 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <16 x i32> [[BROADCAST_SPLATINSERT]], <16 x i32> poison, <16 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[TMP1]], i64 16
-; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP1]], align 1
-; CHECK-NEXT:    store <16 x i8> splat (i8 1), ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 16
+; CHECK-NEXT:    store <16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP1]], align 4
+; CHECK-NEXT:    store <16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP2]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
@@ -38,14 +40,16 @@ define void @test_epilogue_tf(ptr %A, i64 %n) {
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT2]], <8 x i32> poison, <8 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT5:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX2]]
-; CHECK-NEXT:    call void @llvm.masked.store.v8i8.p0(<8 x i8> splat (i8 1), ptr align 1 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX4]]
+; CHECK-NEXT:    call void @llvm.masked.store.v8i32.p0(<8 x i32> [[BROADCAST_SPLAT3]], ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-NEXT:    [[INDEX_NEXT5]] = add i64 [[INDEX4]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT5]], i64 [[N]])
 ; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP6:%.*]] = xor i1 [[TMP5]], true
 ; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
@@ -59,8 +63,8 @@ entry:
 
 for.body:
   %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
-  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
-  store i8 1, ptr %arrayidx, align 1
+  %arrayidx = getelementptr inbounds i32, ptr %A, i64 %iv
+  store i32 %val, ptr %arrayidx, align 4
   %iv.next = add nuw nsw i64 %iv, 1
   %exitcond = icmp ne i64 %iv.next, %n
   br i1 %exitcond, label %for.body, label %exit
@@ -69,6 +73,388 @@ exit:
   ret void
 }
 
+define i32 @add_redc(ptr %src, i64 %n) {
+; CHECK-LABEL: define i32 @add_redc(
+; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i32>, ptr [[TMP3]], align 1
+; CHECK-NEXT:    [[TMP4]] = add <16 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
+; CHECK-NEXT:    [[TMP5]] = add <16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[BIN_RDX:%.*]] = add <16 x i32> [[TMP5]], [[TMP4]]
+; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.add.v16i32(<16 x i32> [[BIN_RDX]])
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i32> [ [[TMP8]], %[[VEC_EPILOG_PH]] ], [ [[TMP11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[TMP9]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
+; CHECK-NEXT:    [[TMP10:%.*]] = add <8 x i32> [[WIDE_MASKED_LOAD]], [[VEC_PHI5]]
+; CHECK-NEXT:    [[TMP11]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP10]], <8 x i32> [[VEC_PHI5]]
+; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP13:%.*]] = xor i1 [[TMP12]], true
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP11]])
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP14]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[ADD_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %red = phi i32 [ 0, %entry ], [ %add, %loop ]
+  %gep = getelementptr inbounds i32, ptr %src, i64 %iv
+  %load = load i32, ptr %gep, align 1
+  %add = add i32 %load, %red
+  %iv.next = add i64 %iv, 1
+  %icmp3 = icmp eq i64 %iv, %n
+  br i1 %icmp3, label %exit, label %loop
+
+exit:
+  ret i32 %add
+}
+
+define i32 @max_redc(ptr %src, i64 %n) {
+; CHECK-LABEL: define i32 @max_redc(
+; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i32>, ptr [[TMP3]], align 1
+; CHECK-NEXT:    [[TMP4]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD]], <16 x i32> [[VEC_PHI]])
+; CHECK-NEXT:    [[TMP5]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD3]], <16 x i32> [[VEC_PHI2]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[RDX_MINMAX:%.*]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[TMP4]], <16 x i32> [[TMP5]])
+; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.umax.v16i32(<16 x i32> [[RDX_MINMAX]])
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i32> poison, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT]], <8 x i32> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i32> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[TMP8]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
+; CHECK-NEXT:    [[TMP9:%.*]] = call <8 x i32> @llvm.umax.v8i32(<8 x i32> [[WIDE_MASKED_LOAD]], <8 x i32> [[VEC_PHI5]])
+; CHECK-NEXT:    [[TMP10]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP9]], <8 x i32> [[VEC_PHI5]]
+; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP12:%.*]] = xor i1 [[TMP11]], true
+; CHECK-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP13:%.*]] = call i32 @llvm.vector.reduce.umax.v8i32(<8 x i32> [[TMP10]])
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP13]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[MAX_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %red = phi i32 [ 0, %entry ], [ %max, %loop ]
+  %gep = getelementptr inbounds i32, ptr %src, i64 %iv
+  %load = load i32, ptr %gep, align 1
+  %max = call i32 @llvm.umax(i32 %load, i32 %red)
+  %iv.next = add i64 %iv, 1
+  %icmp3 = icmp eq i64 %iv, %n
+  br i1 %icmp3, label %exit, label %loop
+
+exit:
+  ret i32 %max
+}
+
+define i32 @live-out(ptr %src, i64 %n) {
+; CHECK-LABEL: define i32 @live-out(
+; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP1]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <16 x i32> [[WIDE_LOAD]], i64 15
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX2]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 4 [[TMP5]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
+; CHECK-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP7:%.*]] = xor i1 [[TMP6]], true
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP8:%.*]] = xor <8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
+; CHECK-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v8i1(<8 x i1> [[TMP8]], i1 false)
+; CHECK-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
+; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <8 x i32> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
+; CHECK-NEXT:    br label %[[FOR_END]]
+; CHECK:       [[FOR_END]]:
+; CHECK-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[TMP9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP4]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[LOAD_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %gep = getelementptr inbounds nuw i32, ptr %src, i64 %iv
+  %load = load i32, ptr %gep, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, %n
+  br i1 %ec, label %for.end, label %loop
+
+for.end:
+  ret i32 %load
+}
+
+define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
+; CHECK-LABEL: define void @reversed-loop(
+; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[ST:%.*]] = sub i32 [[N]], 1
+; CHECK-NEXT:    [[TMP0:%.*]] = add i32 [[N]], -1
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[N]], -2
+; CHECK-NEXT:    [[SMIN1:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP1]], i32 -1)
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[TMP0]], [[SMIN1]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK:       [[VECTOR_SCEVCHECK]]:
+; CHECK-NEXT:    [[TMP3:%.*]] = add i32 [[N]], -2
+; CHECK-NEXT:    [[SMIN:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP3]], i32 -1)
+; CHECK-NEXT:    [[TMP4:%.*]] = sub i32 [[TMP3]], [[SMIN]]
+; CHECK-NEXT:    [[TMP5:%.*]] = sub i32 [[ST]], [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp sgt i32 [[TMP5]], [[ST]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK2:%.*]] = icmp ult i32 [[TMP2]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = and i32 [[TMP2]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP2]], [[TMP7]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <16 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <16 x i32> [[BROADCAST_SPLATINSERT]], <16 x i32> poison, <16 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP8:%.*]] = sub i32 [[ST]], [[N_VEC]]
+; CHECK-NEXT:    [[REVERSE:%.*]] = shufflevector <16 x i32> [[BROADCAST_SPLAT]], <16 x i32> poison, <16 x i32> <i32 15, i32 14, i32 13, i32 12, i32 11, i32 10, i32 9, i32 8, i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP9:%.*]] = sub i32 [[ST]], [[INDEX]]
+; CHECK-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP9]]
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds i32, ptr [[TMP10]], i64 -15
+; CHECK-NEXT:    [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[TMP10]], i64 -31
+; CHECK-NEXT:    store <16 x i32> [[REVERSE]], ptr [[TMP11]], align 4
+; CHECK-NEXT:    store <16 x i32> [[REVERSE]], ptr [[TMP12]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP13:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP2]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT3]], <8 x i32> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[REVERSE5:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLAT4]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[VEC_EPILOG_RESUME_VAL]], i32 [[TMP2]])
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX6:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP14:%.*]] = sub i32 [[ST]], [[INDEX6]]
+; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP14]]
+; CHECK-NEXT:    [[TMP16:%.*]] = getelementptr i32, ptr [[TMP15]], i64 -7
+; CHECK-NEXT:    [[REVERSE7:%.*]] = shufflevector <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    call void @llvm.masked.store.v8i32.p0(<8 x i32> [[REVERSE5]], ptr align 4 [[TMP16]], <8 x i1> [[REVERSE7]])
+; CHECK-NEXT:    [[INDEX_NEXT8]] = add i32 [[INDEX6]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[INDEX_NEXT8]], i32 [[TMP2]])
+; CHECK-NEXT:    [[TMP17:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
+; CHECK-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[ST]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[IV]]
+; CHECK-NEXT:    store i32 [[VAL]], ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = sub nuw nsw i32 [[IV]], 1
+; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp sge i32 [[IV_NEXT]], 0
+; CHECK-NEXT:    br i1 [[EXITCOND]], label %[[FOR_BODY]], label %[[EXIT]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %st = sub i32 %n, 1
+  br label %for.body
+
+for.body:
+  %iv = phi i32 [ %st, %entry ], [ %iv.next, %for.body ]
+  %arrayidx = getelementptr inbounds i32, ptr %A, i32 %iv
+  store i32 %val, ptr %arrayidx, align 4
+  %iv.next = sub nuw nsw i32 %iv, 1
+  %exitcond = icmp sge i32 %iv.next, 0
+  br i1 %exitcond, label %for.body, label %exit
+
+exit:
+  ret void
+}
+
+define void @math_func(ptr %A, i32 %n) {
+; CHECK-LABEL: define void @math_func(
+; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[UMAX:%.*]] = call i32 @llvm.umax.i32(i32 [[N]], i32 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[UMAX]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i32 [[UMAX]], 16
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i32 [[UMAX]], 15
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[UMAX]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds float, ptr [[A]], i32 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x float>, ptr [[TMP1]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = call <16 x float> @llvm.pow.v16f32(<16 x float> [[WIDE_LOAD]], <16 x float> splat (float 2.000000e+00))
+; CHECK-NEXT:    store <16 x float> [[TMP2]], ptr [[TMP1]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 16
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[UMAX]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[VEC_EPILOG_RESUME_VAL]], i32 [[UMAX]])
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX2:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds float, ptr [[A]], i32 [[INDEX2]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x float> @llvm.masked.load.v8f32.p0(ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x float> poison)
+; CHECK-NEXT:    [[TMP5:%.*]] = call <8 x float> @llvm.pow.v8f32(<8 x float> [[WIDE_MASKED_LOAD]], <8 x float> splat (float 2.000000e+00))
+; CHECK-NEXT:    call void @llvm.masked.store.v8f32.p0(<8 x float> [[TMP5]], ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-NEXT:    [[INDEX_NEXT3]] = add i32 [[INDEX2]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[INDEX_NEXT3]], i32 [[UMAX]])
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP7:%.*]] = xor i1 [[TMP6]], true
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP16:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %for.body ]
+  %arrayidx = getelementptr inbounds float, ptr %A, i32 %iv
+  %load = load float, ptr %arrayidx, align 4
+  %val = call float @llvm.pow.f32(float %load, float 2.0)
+  store float %val, ptr %arrayidx, align 4
+  %iv.next = add nuw nsw i32 %iv, 1
+  %exitcond = icmp ult i32 %iv.next, %n
+  br i1 %exitcond, label %for.body, label %exit
+
+exit:
+  ret void
+}
+
 define i64 @test_no_masked_interleave_support(i64 %y, i32 %n) {
 ; CHECK-INVALIDATE-INTERLEAVE-LABEL: Checking a loop in 'test_no_masked_interleave_support'
 ; CHECK-INVALIDATE-INTERLEAVE: LV: epilogue tail-folding is enabled

>From 733c1cd83586405089e290e3ea1a2173d1f33e5b Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Mon, 24 Aug 2026 10:31:47 +0100
Subject: [PATCH 11/25] Give Planner its own epilogue tail-folding CM and make
 plan() responsible for handling epilogueTF planning

---
 .../Vectorize/LoopVectorizationPlanner.h      |  10 +-
 .../Transforms/Vectorize/LoopVectorize.cpp    | 206 +++++----
 .../AArch64/fold-epilogue-tail.ll             | 435 ++++++++++++++----
 3 files changed, 488 insertions(+), 163 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 739c416ed3a5a..291c620bf4b74 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -888,6 +888,10 @@ class LoopVectorizationPlanner {
   /// The profitability analysis. Cleared after making cost based decisions.
   std::unique_ptr<LoopVectorizationCostModel> CM;
 
+  /// The profitability analysis for epilogue tail-folding.
+  /// Cleared after making cost based decisions.
+  std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM;
+
   /// VF selection state independent of cost-modeling decisions.
   VFSelectionContext &Config;
 
@@ -931,6 +935,7 @@ class LoopVectorizationPlanner {
       Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
       const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
       std::unique_ptr<LoopVectorizationCostModel> CM,
+      std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM,
       VFSelectionContext &Config, InterleavedAccessInfo &IAI,
       PredicatedScalarEvolution &PSE, OptimizationRemarkEmitter *ORE,
       std::function<const BranchProbabilityInfo &()> GetBPI);
@@ -946,6 +951,9 @@ class LoopVectorizationPlanner {
   /// Destroy the cost model.
   void clearCostModel();
 
+  /// Destroy the epilogue tail-folding cost model.
+  void clearEpilogueTfCM();
+
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// interleaving should be avoided up-front, no plans are generated.
@@ -953,7 +961,7 @@ class LoopVectorizationPlanner {
 
   /// Build VPlan for the forced epilogue VF. If vectorization and tail-folding
   /// should be avoided up-front, no tail-folded plans are generated.
-  bool planForEpilogueTF(LoopVectorizationCostModel &EpilogueCM);
+  bool planForEpilogueTF();
 
   /// Return the VPlan for \p VF. At the moment, there is always a single VPlan
   /// for each VF.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index de3cc512922c9..7f6a3c0b6b2c1 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3019,7 +3019,10 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
       MaxPowerOf2RuntimeVF = std::nullopt; // Stick with tail-folding for now.
   }
 
-  auto NoScalarEpilogueNeeded = [this, &UserIC](uint64_t MaxRuntimeVF) {
+  // TODO: Make NoScalarEpilogueNeeded lambda a separate function to be used
+  // only for main loop VF not also epilogueVF. Using it for epilogueVF against
+  // full TC is inaccurate.
+  auto NoScalarEpilogueNeeded = [this, &UserIC](unsigned MaxRuntimeVF) {
     // Return false if the loop is neither a single-latch-exit loop nor an
     // early-exit loop as tail-folding is not supported in that case.
     if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
@@ -3164,9 +3167,13 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
   using RecipeVFPair = std::pair<VPRecipeBase *, ElementCount>;
   SmallVector<RecipeVFPair> InvalidCosts;
   for (const auto &Plan : VPlans) {
+    // Skip cost remarks when Plan is not compatible with the CM.
+    // Specifically for the case of epilogue tail-folded Plans.
+    if (Plan->hasTailFolded() ^ CM->preferTailFoldedLoop())
+      continue;
     for (ElementCount VF : Plan->vectorFactors()) {
       // The VPlan-based cost model is designed for computing vector cost.
-      // Querying VPlan-based cost model with a scarlar VF will cause some
+      // Querying VPlan-based cost model with a scalar VF will cause some
       // errors because we expect the VF is vector for most of the widen
       // recipes.
       if (VF.isScalar())
@@ -3394,7 +3401,7 @@ static bool hasFindLastReductionPhi(VPlan &Plan) {
 /// otherwise CM_EpilogueAllowed.
 static EpilogueLowering getEpilogueTailLowering(
     const LoopVectorizationCostModel &MainCM, const Loop *L,
-    OptimizationRemarkEmitter *ORE, LoopVectorizationLegality &LVL,
+    OptimizationRemarkEmitter *ORE, const LoopVectorizationLegality &LVL,
     const LoopVectorizeHints &Hints, TargetTransformInfo *TTI) {
   // Epilogue TF is only enabled when explicitly requested via command line.
   if (!EpilogueTailFoldingPolicy.getNumOccurrences() ||
@@ -5410,21 +5417,27 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
       // end.
       buildVPlans(*VPlan1, UserVF, UserVF, *CM);
 
-      ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
-      if (EpilogueUserVF.isVector() &&
-          ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
-        CM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
-        buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, *CM);
-      }
-      if (!VPlans.empty() && VPlans.front()->getSingleVF() == UserVF) {
-        // For scalar VF, skip VPlan cost check as VPlan cost is designed for
-        // vector VFs only.
-        if (UserVF.isScalar() ||
-            cost(*VPlans.front(), UserVF, /*RU=*/nullptr, *CM).isValid()) {
-          LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
-          LLVM_DEBUG(printPlans(dbgs()));
-          return;
+      // For scalar VF, skip VPlan cost check as VPlan cost is designed for
+      // vector VFs only.
+      if (!VPlans.empty() &&
+          (UserVF.isScalar() ||
+           cost(*VPlans.front(), UserVF, /*RU=*/nullptr, *CM).isValid())) {
+        // Plan for epilogue only if we succeeded in building main loop vplan.
+
+        // Try to plan for tail-folded epilogue if it's enabled/doable,
+        // otherwise plan for unpredicated epilogue:
+        bool EpilogueTfPlanCreated = planForEpilogueTF();
+        if (!EpilogueTfPlanCreated) {
+          ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
+          if (EpilogueUserVF.isVector() &&
+              ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
+            CM->collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
+            buildVPlans(*VPlan1, EpilogueUserVF, EpilogueUserVF, *CM);
+          }
         }
+        LLVM_DEBUG(dbgs() << "LV: Using user VF " << UserVF << ".\n");
+        LLVM_DEBUG(printPlans(dbgs()));
+        return;
       }
       VPlans.clear();
       reportVectorizationInfo("UserVF ignored because of invalid costs.",
@@ -5453,61 +5466,80 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   LLVM_DEBUG(printPlans(dbgs()));
 }
 
-bool LoopVectorizationPlanner::planForEpilogueTF(
-    LoopVectorizationCostModel &EpilogueCM) {
-  if (VPlans.empty()) {
-    LLVM_DEBUG(dbgs() << "LV: no vplans have been built for main loop VF, bail "
-                         "out of epilogue tail-folding\n");
+bool LoopVectorizationPlanner::planForEpilogueTF() {
+  if (!EpilogueTfCM)
     return false;
-  }
+  assert(EpilogueTfCM->preferTailFoldedLoop() &&
+         "Epilogue tail-folding is expected to be enabled");
+
+  LLVM_DEBUG(dbgs() << "LV: plan for tail-folded epilogue\n");
 
-  EpilogueCM.ValuesToIgnore.insert_range(CM->ValuesToIgnore);
-  EpilogueCM.VecValuesToIgnore.insert_range(CM->VecValuesToIgnore);
+  EpilogueTfCM->ValuesToIgnore.insert_range(CM->ValuesToIgnore);
+  EpilogueTfCM->VecValuesToIgnore.insert_range(CM->VecValuesToIgnore);
 
   FixedScalableVFPair MaxFactors =
-      EpilogueCM.computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
-  if (!MaxFactors ||
-      !EpilogueCM
-           .foldTailByMasking()) { // Cases that should not to be vectorized
-                                   // or tail-folded.
+      EpilogueTfCM->computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
+  if (!MaxFactors || !EpilogueTfCM->foldTailByMasking()) {
+    // Cases that should not to be vectorized or tail-folded.
     reportVectorizationInfo("This case of epilogue loop can't be tail-folded",
                             "InvalidTailFoldedEpilogue", ORE, OrigLoop);
     return false;
   }
 
-  auto VPlan1 = tryToBuildVPlan1(EpilogueCM);
+  auto VPlan1 = tryToBuildVPlan1(*EpilogueTfCM);
+
+  // If we're here, the main loop's initial VPlan was built successfully.
+  // Building one for the tail-folded loop should therefore also succeed, since
+  // nothing tail-folding-specific happens yet at this point. Still check below
+  // to catch any unexpected failure.
+  if (!VPlan1) {
+    reportVectorizationInfo(
+        "Failed to build initial tail-folded epilogue VPlan",
+        "InvalidTailFoldedEpilogue", ORE, OrigLoop);
+    return false;
+  }
 
   if (!useMaskedInterleavedAccesses(TTI)) {
     LLVM_DEBUG(
         dbgs() << "LV: Invalidate all interleaved groups due to fold-tail by "
                   "masking which requires masked-interleaved support.\n");
-    if (EpilogueCM.InterleaveInfo.invalidateGroups())
+    if (EpilogueTfCM->InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      EpilogueCM.invalidateCostModelingDecisions();
+      EpilogueTfCM->invalidateCostModelingDecisions();
   }
   Legal->prepareToFoldTailByMasking();
 
   // Collect the instructions (and their associated costs) that will be more
   // profitable to scalarize.
-  EpilogueCM.collectNonVectorizedAndSetWideningDecisions(
+  EpilogueTfCM->collectNonVectorizedAndSetWideningDecisions(
       EpilogueVectorizationForceVF);
 
-  // Expecting only 2 vplans as epilogue TF is applied only for forced VFs.
-  assert(VPlans.size() == 2 &&
-         "For tail-folded epilogue, VPlans size is expected to be 2");
-  // Remove the last vplan, which should be the epilogue plan to replace it by
-  // the tail-folded vplan:
-  assert(VPlans.back()->getSingleVF() == EpilogueVectorizationForceVF &&
-         "For tail-folded epilogue, last vplan is expected to have "
-         "EpilogueUserVF");
-  VPlans.pop_back();
+  size_t NumPlansBefore = VPlans.size();
   buildVPlans(*VPlan1, EpilogueVectorizationForceVF,
-              EpilogueVectorizationForceVF, EpilogueCM);
+              EpilogueVectorizationForceVF, *EpilogueTfCM);
+
+  // Check that a vplan is successfully built:
+  if (VPlans.size() == NumPlansBefore ||
+      VPlans.back()->getSingleVF() != EpilogueVectorizationForceVF ||
+      !VPlans.back()->hasTailFolded()) {
+    reportVectorizationInfo(
+        "Failed to build a valid tail-folded epilogue VPlan",
+        "InvalidTailFoldedEpilogue", ORE, OrigLoop);
+    return false;
+  }
 
-  cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr,
-       EpilogueCM);
+  if (!cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr,
+            *EpilogueTfCM)
+           .isValid()) {
+    VPlans.pop_back();
+    reportVectorizationInfo("This case of epilogue loop can't be tail-folded "
+                            "- Invalid costs",
+                            "InvalidTailFoldedEpilogue", ORE, OrigLoop);
+    return false;
+  }
+  LLVM_DEBUG(dbgs() << "LV: Tail-folded epilogue VPlan is created\n");
   return true;
 }
 
@@ -5817,18 +5849,21 @@ LoopVectorizationPlanner::computeBestVF() {
 LoopVectorizationPlanner::LoopVectorizationPlanner(
     Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
     const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
-    std::unique_ptr<LoopVectorizationCostModel> CM, VFSelectionContext &Config,
-    InterleavedAccessInfo &IAI, PredicatedScalarEvolution &PSE,
-    OptimizationRemarkEmitter *ORE,
+    std::unique_ptr<LoopVectorizationCostModel> CM,
+    std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM,
+    VFSelectionContext &Config, InterleavedAccessInfo &IAI,
+    PredicatedScalarEvolution &PSE, OptimizationRemarkEmitter *ORE,
     std::function<const BranchProbabilityInfo &()> GetBPI)
     : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal),
-      CM(std::move(CM)), Config(Config), IAI(IAI), PSE(PSE), ORE(ORE),
-      GetBPI(GetBPI) {}
+      CM(std::move(CM)), EpilogueTfCM(std::move(EpilogueTfCM)), Config(Config),
+      IAI(IAI), PSE(PSE), ORE(ORE), GetBPI(GetBPI) {}
 
 LoopVectorizationPlanner::~LoopVectorizationPlanner() = default;
 
 void LoopVectorizationPlanner::clearCostModel() { CM.reset(); }
 
+void LoopVectorizationPlanner::clearEpilogueTfCM() { EpilogueTfCM.reset(); }
+
 DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
     ElementCount BestVF, unsigned BestUF, VPlan &BestVPlan,
     InnerLoopVectorizer &ILV, DominatorTree *DT,
@@ -5856,8 +5891,9 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
                    BestVPlan, BestVF, VScale);
   }
 
+  const bool IsTailFolded = BestVPlan.hasTailFolded();
   if (vputils::findIncomingAliasMask(BestVPlan)) {
-    assert(BestVPlan.hasTailFolded() && "Expected tail folding to be enabled");
+    assert(IsTailFolded && "Expected tail folding to be enabled");
     RUN_VPLAN_PASS(VPlanTransforms::materializeAliasMaskCheckBlock, BestVPlan,
                    *Legal->getRuntimePointerChecking()->getDiffChecks(),
                    HasBranchWeights);
@@ -5897,7 +5933,6 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
   RUN_VPLAN_PASS(VPlanTransforms::convertEVLExitCond, BestVPlan);
   // Regions are dissolved after optimizing for VF and UF, which completely
   // removes unneeded loop regions first.
-  const bool HasTailFolded = BestVPlan.hasTailFolded();
   RUN_VPLAN_PASS(VPlanTransforms::dissolveLoopRegions, BestVPlan);
   // Expand BranchOnTwoConds after dissolution, when latch has direct access to
   // its successors.
@@ -5917,7 +5952,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
   assert((LI->getUniqueLatchExitBlock(*OrigLoop) || RequiresScalarEpilogue) &&
          "loops not exiting via the latch without required epilogue?");
   RUN_VPLAN_PASS(VPlanTransforms::materializeVectorTripCount, BestVPlan,
-                 VectorPH, HasTailFolded, RequiresScalarEpilogue,
+                 VectorPH, IsTailFolded, RequiresScalarEpilogue,
                  &BestVPlan.getVFxUF(), MaxRuntimeStep);
   RUN_VPLAN_PASS(VPlanTransforms::materializeFactors, BestVPlan, VectorPH,
                  BestVF);
@@ -7724,7 +7759,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
                                       VPIRBasicBlock *VecEpilogueIterCheckVPBB,
                                       ArrayRef<Instruction *> InstsToMove,
                                       ArrayRef<VPInstruction *> ResumeValues,
-                                      bool IsEpilogueTFEnabled) {
+                                      bool IsEpilogueTfEnabled) {
   ArrayRef<VPBlockBase *> Preds = VecEpilogueIterCheckVPBB->getPredecessors();
   BasicBlock *MainLoopIterationCountCheck =
       cast<VPIRBasicBlock>(Preds.front())->getIRBasicBlock();
@@ -7751,7 +7786,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
   // the epilogue plan.
   BasicBlock *EpilogueIterationCountCheck =
       cast<VPIRBasicBlock>(EpiPlan.getEntry())->getIRBasicBlock();
-  if (IsEpilogueTFEnabled) {
+  if (IsEpilogueTfEnabled) {
     // With a tail-folded epilogue there is no scalar remainder to bail
     // to, even a trip count too small for the epilogue VF is handled safely by
     // the masked epilogue vector loop, so skip straight to its preheader.
@@ -7781,13 +7816,13 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
         VecEpilogueIterationCountCheck);
     // When the epilogue is tail-folded, EpilogueIterationCountCheck
     // (iter.check) is redirected to branch straight into the vector epilogue
-    // preheader (see the IsEpilogueTFEnabled redirect above), so it is now a
+    // preheader (see the IsEpilogueTfEnabled redirect above), so it is now a
     // genuine predecessor. Like MainLoopIterationCountCheck, it bypasses the
     // main vector loop, so re-use the incoming value from that edge.
     // TODO: revisit for reduction phis, whose resume value on this bypass
     // edge may need dedicated handling rather than reusing the value already
     // present here.
-    if (IsEpilogueTFEnabled)
+    if (IsEpilogueTfEnabled)
       Phi->addIncoming(
           Phi->getIncomingValueForBlock(MainLoopIterationCountCheck),
           EpilogueIterationCountCheck);
@@ -7809,7 +7844,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
     if (Phi.use_empty())
       Phi.eraseFromParent();
 
-  if (IsEpilogueTFEnabled) {
+  if (IsEpilogueTfEnabled) {
     // The epilogue vector loop is tail-folded, so it can safely handle
     // any remaining iterations, including zero, via masking.
     // vec.epilog.iter.check's own min-iters check was therefore built with a
@@ -8028,28 +8063,34 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Use the cost model.
   VFSelectionContext Config(*TTI, &LVL, L, *F, PSE, DB, ORE, &Hints,
                             OptForSize);
-  // Use the planner for vectorization.
-  LoopVectorizationPlanner LVP(
-      L, LI, DT, TLI, *TTI, &LVL,
-      std::make_unique<LoopVectorizationCostModel>(
-          SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config),
-      Config, IAI, PSE, ORE, GetBPI);
-
+  auto CM = std::make_unique<LoopVectorizationCostModel>(
+      SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config);
+
+  // Setup the epilogue tail-folding CM. Only built when tail-folding the
+  // epilogue is actually a candidate, to avoid the cost of an extra
+  // InterleavedAccessInfo scan and LoopVectorizationCostModel construction
+  // for the common case where this (experimental, off-by-default) feature
+  // isn't in use.
   EpilogueLowering EpilogueTailLoweringStatus =
-      getEpilogueTailLowering(LVP.getCostModel(), L, ORE, LVL, Hints, TTI);
-  std::optional<InterleavedAccessInfo> TailFoldingCMIAI;
-  std::optional<LoopVectorizationCostModel> EpilogueTailFoldingCM;
+      getEpilogueTailLowering(*CM, L, ORE, LVL, Hints, TTI);
+  std::optional<InterleavedAccessInfo> EpilogueTfCMIAI;
+  std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM;
   if (EpilogueTailLoweringStatus ==
       EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
     LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
-    TailFoldingCMIAI.emplace(PSE, L, DT, LI, LVL.getLAI(), OptForSize);
+    EpilogueTfCMIAI.emplace(PSE, L, DT, LI, LVL.getLAI(), OptForSize);
     if (UseInterleaved)
-      TailFoldingCMIAI->analyzeInterleaving(useMaskedInterleavedAccesses(*TTI));
-    EpilogueTailFoldingCM.emplace(CM_EpilogueNotNeededFoldTail, L, PSE, LI,
-                                  &LVL, *TTI, TLI, AC, ORE, GetBFI, F,
-                                  *TailFoldingCMIAI, Config);
+      EpilogueTfCMIAI->analyzeInterleaving(useMaskedInterleavedAccesses(*TTI));
+    EpilogueTfCM = std::make_unique<LoopVectorizationCostModel>(
+        EpilogueTailLoweringStatus, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE,
+        GetBFI, F, *EpilogueTfCMIAI, Config);
   }
 
+  // Use the planner for vectorization.
+  LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, std::move(CM),
+                               std::move(EpilogueTfCM), Config, IAI, PSE,
+                               ORE, GetBPI);
+
   // Get user vectorization factor and interleave count.
   ElementCount UserVF = Hints.getWidth();
   unsigned UserIC = Hints.getInterleave();
@@ -8062,14 +8103,9 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC);
-  if (EpilogueTailFoldingCM)
-    if (!LVP.planForEpilogueTF(EpilogueTailFoldingCM.value())) {
-      // we can't apply epilogue TF:
-      reportVectorizationInfo(
-          "Applying epilogue tail-folding failed, disable it.",
-          "InvalidTailFoldedEpilogue", ORE, L);
-      EpilogueTailFoldingCM.reset();
-    }
+  // Right now, after planning, the epilogue tail-folding CM is not needed
+  // anymore. Clear it.
+  LVP.clearEpilogueTfCM();
 
   auto [VF, BestPlanPtr] = LVP.computeBestVF();
   unsigned IC = 1;
@@ -8330,14 +8366,16 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     SmallVector<Instruction *> InstsToMove = preparePlanForEpilogueVectorLoop(
         BestMainPlan, BestEpiPlan, L, ExpandedSCEVs, EPI, LVP, Config,
         *PSE.getSE(), ResumeValues);
+    LVP.attachRuntimeChecks(BestEpiPlan, Checks, HasBranchWeights);
     RUN_VPLAN_PASS(VPlanTransforms::simplifyLiveInsWithSCEV, BestEpiPlan, PSE);
+    // Save the status of epilogue tail-folding:
+    const bool IsTailFolded = BestEpiPlan.hasTailFolded();
     LVP.executePlan(
         EPI.EpilogueVF, EPI.EpilogueUF, BestEpiPlan, EpilogILV, DT,
         LoopVectorizationPlanner::EpilogueVectorizationKind::Epilogue);
     connectEpilogueVectorLoop(BestEpiPlan, L, DT, LI, Checks,
                               EpilogILV.VecEpilogueIterationCountCheck,
-                              InstsToMove, ResumeValues,
-                              EpilogueTailFoldingCM.has_value());
+                              InstsToMove, ResumeValues, IsTailFolded);
     ++LoopsEpilogueVectorized;
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 79495c4ad9c3e..ffde0d0084933 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,7 +1,10 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; REQUIRES: asserts
 ; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug-only=loop-vectorize -mcpu=neoverse-v1 -S %s | FileCheck %s
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug-only=loop-vectorize -mattr=+sve -S %s | FileCheck %s
+
+; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width="vscale x 16" -epilogue-vectorization-force-VF="vscale x 8" -debug-only=loop-vectorize -mattr=+sve -S %s | FileCheck %s --check-prefix=CHECK-VS
 
 ; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize,vectorutils --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
 ; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALIDATE-INTERLEAVE
@@ -31,7 +34,7 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-NEXT:    store <16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP2]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
@@ -52,12 +55,68 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT5]], i64 [[N]])
 ; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP6:%.*]] = xor i1 [[TMP5]], true
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    br label %[[EXIT]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
+; CHECK-VS-LABEL: define void @test_epilogue_tf(
+; CHECK-VS-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP4]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 [[TMP3]]
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP5]], align 4
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP6]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = shl nuw i64 [[TMP8]], 3
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT2]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT5:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX4]]
+; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[BROADCAST_SPLAT3]], ptr align 4 [[TMP10]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT5]] = add i64 [[INDEX4]], [[TMP9]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT5]], i64 [[N]])
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = xor i1 [[TMP11]], true
+; CHECK-VS-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    ret void
+;
 entry:
   br label %for.body
 
@@ -99,7 +158,7 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[TMP5]] = add <16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[BIN_RDX:%.*]] = add <16 x i32> [[TMP5]], [[TMP4]]
 ; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.add.v16i32(<16 x i32> [[BIN_RDX]])
@@ -125,7 +184,7 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
 ; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP13:%.*]] = xor i1 [[TMP12]], true
-; CHECK-NEXT:    br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP11]])
 ; CHECK-NEXT:    br label %[[EXIT]]
@@ -133,6 +192,72 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP14]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[ADD_LCSSA]]
 ;
+; CHECK-VS-LABEL: define i32 @add_redc(
+; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP4]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 1
+; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
+; CHECK-VS-NEXT:    [[TMP8]] = add <vscale x 16 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
+; CHECK-VS-NEXT:    [[TMP9]] = add <vscale x 16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[BIN_RDX:%.*]] = add <vscale x 16 x i32> [[TMP9]], [[TMP8]]
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i32 @llvm.vector.reduce.add.nxv16i32(<vscale x 16 x i32> [[BIN_RDX]])
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = insertelement <vscale x 8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i32> [ [[TMP14]], %[[VEC_EPILOG_PH]] ], [ [[TMP17:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 1 [[TMP15]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = add <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], [[VEC_PHI5]]
+; CHECK-VS-NEXT:    [[TMP17]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> [[TMP16]], <vscale x 8 x i32> [[VEC_PHI5]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP13]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = xor i1 [[TMP18]], true
+; CHECK-VS-NEXT:    br i1 [[TMP19]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP20:%.*]] = call i32 @llvm.vector.reduce.add.nxv8i32(<vscale x 8 x i32> [[TMP17]])
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP20]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i32 [[ADD_LCSSA]]
+;
 entry:
   br label %loop
 
@@ -176,7 +301,7 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[TMP5]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD3]], <16 x i32> [[VEC_PHI2]])
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[RDX_MINMAX:%.*]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[TMP4]], <16 x i32> [[TMP5]])
 ; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.umax.v16i32(<16 x i32> [[RDX_MINMAX]])
@@ -203,7 +328,7 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
 ; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP12:%.*]] = xor i1 [[TMP11]], true
-; CHECK-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[TMP13:%.*]] = call i32 @llvm.vector.reduce.umax.v8i32(<8 x i32> [[TMP10]])
 ; CHECK-NEXT:    br label %[[EXIT]]
@@ -211,6 +336,73 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP13]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[MAX_LCSSA]]
 ;
+; CHECK-VS-LABEL: define i32 @max_redc(
+; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP4]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 1
+; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
+; CHECK-VS-NEXT:    [[TMP8]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD]], <vscale x 16 x i32> [[VEC_PHI]])
+; CHECK-VS-NEXT:    [[TMP9]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], <vscale x 16 x i32> [[VEC_PHI2]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[RDX_MINMAX:%.*]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[TMP8]], <vscale x 16 x i32> [[TMP9]])
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i32 @llvm.vector.reduce.umax.nxv16i32(<vscale x 16 x i32> [[RDX_MINMAX]])
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i32> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP16:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 1 [[TMP14]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = call <vscale x 8 x i32> @llvm.umax.nxv8i32(<vscale x 8 x i32> [[WIDE_MASKED_LOAD]], <vscale x 8 x i32> [[VEC_PHI5]])
+; CHECK-VS-NEXT:    [[TMP16]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> [[TMP15]], <vscale x 8 x i32> [[VEC_PHI5]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP13]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
+; CHECK-VS-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = call i32 @llvm.vector.reduce.umax.nxv8i32(<vscale x 8 x i32> [[TMP16]])
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP19]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i32 [[MAX_LCSSA]]
+;
 entry:
   br label %loop
 
@@ -248,7 +440,7 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <16 x i32> [[WIDE_LOAD]], i64 15
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
@@ -268,7 +460,7 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
 ; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP7:%.*]] = xor i1 [[TMP6]], true
-; CHECK-NEXT:    br i1 [[TMP7]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[TMP8:%.*]] = xor <8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
 ; CHECK-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v8i1(<8 x i1> [[TMP8]], i1 false)
@@ -279,6 +471,66 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[TMP9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP4]], %[[MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[LOAD_LCSSA]]
 ;
+; CHECK-VS-LABEL: define i32 @live-out(
+; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP4]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP5]], i64 [[TMP3]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = mul nuw i32 [[TMP8]], 16
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = sub i32 [[TMP9]], 1
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = extractelement <vscale x 16 x i32> [[WIDE_LOAD]], i32 [[TMP10]]
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX2]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 4 [[TMP14]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], [[TMP13]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = xor i1 [[TMP15]], true
+; CHECK-VS-NEXT:    br i1 [[TMP16]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = xor <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
+; CHECK-VS-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP17]], i1 false)
+; CHECK-VS-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = extractelement <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
+; CHECK-VS-NEXT:    br label %[[FOR_END]]
+; CHECK-VS:       [[FOR_END]]:
+; CHECK-VS-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[TMP18]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i32 [[LOAD_LCSSA]]
+;
 entry:
   br label %loop
 
@@ -333,7 +585,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    store <16 x i32> [[REVERSE]], ptr [[TMP12]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP13:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP2]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
@@ -358,7 +610,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[INDEX_NEXT8]], i32 [[TMP2]])
 ; CHECK-NEXT:    [[TMP17:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
-; CHECK-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    br label %[[EXIT]]
 ; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
@@ -369,10 +621,102 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    store i32 [[VAL]], ptr [[ARRAYIDX]], align 4
 ; CHECK-NEXT:    [[IV_NEXT]] = sub nuw nsw i32 [[IV]], 1
 ; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp sge i32 [[IV_NEXT]], 0
-; CHECK-NEXT:    br i1 [[EXITCOND]], label %[[FOR_BODY]], label %[[EXIT]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK-NEXT:    br i1 [[EXITCOND]], label %[[FOR_BODY]], label %[[EXIT]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
+; CHECK-VS-LABEL: define void @reversed-loop(
+; CHECK-VS-SAME: ptr [[A:%.*]], i32 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[ST:%.*]] = sub i32 [[N]], 1
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i32 [[N]], -1
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = add i32 [[N]], -2
+; CHECK-VS-NEXT:    [[SMIN1:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP1]], i32 -1)
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = sub i32 [[TMP0]], [[SMIN1]]
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i32 [[TMP3]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], [[TMP4]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK-VS:       [[VECTOR_SCEVCHECK]]:
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = add i32 [[N]], -2
+; CHECK-VS-NEXT:    [[SMIN:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP5]], i32 -1)
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = sub i32 [[TMP5]], [[SMIN]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = sub i32 [[ST]], [[TMP6]]
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = icmp sgt i32 [[TMP7]], [[ST]]
+; CHECK-VS-NEXT:    br i1 [[TMP8]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = shl nuw i32 [[TMP3]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK2:%.*]] = icmp ult i32 [[TMP2]], [[TMP9]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = shl nuw i32 [[TMP3]], 4
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = shl nuw i32 [[TMP10]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i32 [[TMP2]], [[TMP11]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP2]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = sub i32 [[ST]], [[N_VEC]]
+; CHECK-VS-NEXT:    [[REVERSE:%.*]] = call <vscale x 16 x i32> @llvm.vector.reverse.nxv16i32(<vscale x 16 x i32> [[BROADCAST_SPLAT]])
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = sub i32 [[ST]], [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP13]]
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = zext i32 [[TMP10]] to i64
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = sub nuw nsw i64 [[TMP15]], 1
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = sub i64 0, [[TMP16]]
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[TMP14]], i64 [[TMP17]]
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = sub i64 [[TMP17]], [[TMP15]]
+; CHECK-VS-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[TMP14]], i64 [[TMP19]]
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP18]], align 4
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP20]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], [[TMP11]]
+; CHECK-VS-NEXT:    [[TMP21:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP21]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP2]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP23:%.*]] = shl nuw i32 [[TMP22]], 3
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT3]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[REVERSE5:%.*]] = call <vscale x 8 x i32> @llvm.vector.reverse.nxv8i32(<vscale x 8 x i32> [[BROADCAST_SPLAT4]])
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i32(i32 [[VEC_EPILOG_RESUME_VAL]], i32 [[TMP2]])
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX6:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP24:%.*]] = sub i32 [[ST]], [[INDEX6]]
+; CHECK-VS-NEXT:    [[TMP25:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP24]]
+; CHECK-VS-NEXT:    [[TMP26:%.*]] = zext i32 [[TMP23]] to i64
+; CHECK-VS-NEXT:    [[TMP27:%.*]] = sub nuw nsw i64 [[TMP26]], 1
+; CHECK-VS-NEXT:    [[TMP28:%.*]] = sub i64 0, [[TMP27]]
+; CHECK-VS-NEXT:    [[TMP29:%.*]] = getelementptr i32, ptr [[TMP25]], i64 [[TMP28]]
+; CHECK-VS-NEXT:    [[REVERSE7:%.*]] = call <vscale x 8 x i1> @llvm.vector.reverse.nxv8i1(<vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[REVERSE5]], ptr align 4 [[TMP29]], <vscale x 8 x i1> [[REVERSE7]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT8]] = add i32 [[INDEX6]], [[TMP23]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i32(i32 [[INDEX_NEXT8]], i32 [[TMP2]])
+; CHECK-VS-NEXT:    [[TMP30:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP31:%.*]] = xor i1 [[TMP30]], true
+; CHECK-VS-NEXT:    br i1 [[TMP31]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-VS-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-VS:       [[FOR_BODY]]:
+; CHECK-VS-NEXT:    [[IV:%.*]] = phi i32 [ [[ST]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[IV]]
+; CHECK-VS-NEXT:    store i32 [[VAL]], ptr [[ARRAYIDX]], align 4
+; CHECK-VS-NEXT:    [[IV_NEXT]] = sub nuw nsw i32 [[IV]], 1
+; CHECK-VS-NEXT:    [[EXITCOND:%.*]] = icmp sge i32 [[IV_NEXT]], 0
+; CHECK-VS-NEXT:    br i1 [[EXITCOND]], label %[[FOR_BODY]], label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    ret void
+;
 entry:
   %st = sub i32 %n, 1
   br label %for.body
@@ -389,72 +733,6 @@ exit:
   ret void
 }
 
-define void @math_func(ptr %A, i32 %n) {
-; CHECK-LABEL: define void @math_func(
-; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-NEXT:    [[UMAX:%.*]] = call i32 @llvm.umax.i32(i32 [[N]], i32 1)
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[UMAX]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i32 [[UMAX]], 16
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = and i32 [[UMAX]], 15
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[UMAX]], [[TMP0]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds float, ptr [[A]], i32 [[INDEX]]
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x float>, ptr [[TMP1]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = call <16 x float> @llvm.pow.v16f32(<16 x float> [[WIDE_LOAD]], <16 x float> splat (float 2.000000e+00))
-; CHECK-NEXT:    store <16 x float> [[TMP2]], ptr [[TMP1]], align 4
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 16
-; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[UMAX]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
-; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[VEC_EPILOG_RESUME_VAL]], i32 [[UMAX]])
-; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX2:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds float, ptr [[A]], i32 [[INDEX2]]
-; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x float> @llvm.masked.load.v8f32.p0(ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x float> poison)
-; CHECK-NEXT:    [[TMP5:%.*]] = call <8 x float> @llvm.pow.v8f32(<8 x float> [[WIDE_MASKED_LOAD]], <8 x float> splat (float 2.000000e+00))
-; CHECK-NEXT:    call void @llvm.masked.store.v8f32.p0(<8 x float> [[TMP5]], ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-NEXT:    [[INDEX_NEXT3]] = add i32 [[INDEX2]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[INDEX_NEXT3]], i32 [[UMAX]])
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-NEXT:    [[TMP7:%.*]] = xor i1 [[TMP6]], true
-; CHECK-NEXT:    br i1 [[TMP7]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP16:![0-9]+]]
-; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-NEXT:    br label %[[EXIT]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    ret void
-;
-entry:
-  br label %for.body
-
-for.body:
-  %iv = phi i32 [ 0, %entry ], [ %iv.next, %for.body ]
-  %arrayidx = getelementptr inbounds float, ptr %A, i32 %iv
-  %load = load float, ptr %arrayidx, align 4
-  %val = call float @llvm.pow.f32(float %load, float 2.0)
-  store float %val, ptr %arrayidx, align 4
-  %iv.next = add nuw nsw i32 %iv, 1
-  %exitcond = icmp ult i32 %iv.next, %n
-  br i1 %exitcond, label %for.body, label %exit
-
-exit:
-  ret void
-}
-
 define i64 @test_no_masked_interleave_support(i64 %y, i32 %n) {
 ; CHECK-INVALIDATE-INTERLEAVE-LABEL: Checking a loop in 'test_no_masked_interleave_support'
 ; CHECK-INVALIDATE-INTERLEAVE: LV: epilogue tail-folding is enabled
@@ -481,3 +759,4 @@ cond.end:
 for.cond.cleanup:
   ret i64 %cond
 }
+

>From 37bb182e20d6c5ad5143e5316635b0b07d6eace7 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Mon, 24 Aug 2026 20:21:28 +0100
Subject: [PATCH 12/25] Add supported reduction cases

---
 .../AArch64/fold-epilogue-tail-reductions.ll  | 584 ++++++++++++++++++
 .../AArch64/fold-epilogue-tail.ll             | 288 ---------
 2 files changed, 584 insertions(+), 288 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll
new file mode 100644
index 0000000000000..1b2088b617606
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll
@@ -0,0 +1,584 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; REQUIRES: asserts
+; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -mattr=+sve -S %s | FileCheck %s
+
+; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width="vscale x 16" -epilogue-vectorization-force-VF="vscale x 8" -mattr=+sve -S %s | FileCheck %s --check-prefix=CHECK-VS
+
+target triple = "aarch64-linux-gnu"
+
+define i32 @add_redc(ptr %src, i64 %n) {
+; CHECK-LABEL: define i32 @add_redc(
+; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i32>, ptr [[TMP3]], align 1
+; CHECK-NEXT:    [[TMP4]] = add <16 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
+; CHECK-NEXT:    [[TMP5]] = add <16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[BIN_RDX:%.*]] = add <16 x i32> [[TMP5]], [[TMP4]]
+; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.add.v16i32(<16 x i32> [[BIN_RDX]])
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i32> [ [[TMP8]], %[[VEC_EPILOG_PH]] ], [ [[TMP11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[TMP9]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
+; CHECK-NEXT:    [[TMP10:%.*]] = add <8 x i32> [[WIDE_MASKED_LOAD]], [[VEC_PHI5]]
+; CHECK-NEXT:    [[TMP11]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP10]], <8 x i32> [[VEC_PHI5]]
+; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP13:%.*]] = xor i1 [[TMP12]], true
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP11]])
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP14]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[ADD_LCSSA]]
+;
+; CHECK-VS-LABEL: define i32 @add_redc(
+; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP4]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 1
+; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
+; CHECK-VS-NEXT:    [[TMP8]] = add <vscale x 16 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
+; CHECK-VS-NEXT:    [[TMP9]] = add <vscale x 16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[BIN_RDX:%.*]] = add <vscale x 16 x i32> [[TMP9]], [[TMP8]]
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i32 @llvm.vector.reduce.add.nxv16i32(<vscale x 16 x i32> [[BIN_RDX]])
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = insertelement <vscale x 8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i32> [ [[TMP14]], %[[VEC_EPILOG_PH]] ], [ [[TMP17:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 1 [[TMP15]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = add <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], [[VEC_PHI5]]
+; CHECK-VS-NEXT:    [[TMP17]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> [[TMP16]], <vscale x 8 x i32> [[VEC_PHI5]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP13]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = xor i1 [[TMP18]], true
+; CHECK-VS-NEXT:    br i1 [[TMP19]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP20:%.*]] = call i32 @llvm.vector.reduce.add.nxv8i32(<vscale x 8 x i32> [[TMP17]])
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP20]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i32 [[ADD_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %red = phi i32 [ 0, %entry ], [ %add, %loop ]
+  %gep = getelementptr inbounds i32, ptr %src, i64 %iv
+  %load = load i32, ptr %gep, align 1
+  %add = add i32 %load, %red
+  %iv.next = add i64 %iv, 1
+  %icmp3 = icmp eq i64 %iv, %n
+  br i1 %icmp3, label %exit, label %loop
+
+exit:
+  ret i32 %add
+}
+
+define i32 @max_redc(ptr %src, i64 %n) {
+; CHECK-LABEL: define i32 @max_redc(
+; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i32>, ptr [[TMP3]], align 1
+; CHECK-NEXT:    [[TMP4]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD]], <16 x i32> [[VEC_PHI]])
+; CHECK-NEXT:    [[TMP5]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD3]], <16 x i32> [[VEC_PHI2]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[RDX_MINMAX:%.*]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[TMP4]], <16 x i32> [[TMP5]])
+; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.umax.v16i32(<16 x i32> [[RDX_MINMAX]])
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i32> poison, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT]], <8 x i32> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i32> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[TMP8]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
+; CHECK-NEXT:    [[TMP9:%.*]] = call <8 x i32> @llvm.umax.v8i32(<8 x i32> [[WIDE_MASKED_LOAD]], <8 x i32> [[VEC_PHI5]])
+; CHECK-NEXT:    [[TMP10]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP9]], <8 x i32> [[VEC_PHI5]]
+; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP12:%.*]] = xor i1 [[TMP11]], true
+; CHECK-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP13:%.*]] = call i32 @llvm.vector.reduce.umax.v8i32(<8 x i32> [[TMP10]])
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP13]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[MAX_LCSSA]]
+;
+; CHECK-VS-LABEL: define i32 @max_redc(
+; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP4]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 1
+; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
+; CHECK-VS-NEXT:    [[TMP8]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD]], <vscale x 16 x i32> [[VEC_PHI]])
+; CHECK-VS-NEXT:    [[TMP9]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], <vscale x 16 x i32> [[VEC_PHI2]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[RDX_MINMAX:%.*]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[TMP8]], <vscale x 16 x i32> [[TMP9]])
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i32 @llvm.vector.reduce.umax.nxv16i32(<vscale x 16 x i32> [[RDX_MINMAX]])
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i32> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP16:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 1 [[TMP14]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = call <vscale x 8 x i32> @llvm.umax.nxv8i32(<vscale x 8 x i32> [[WIDE_MASKED_LOAD]], <vscale x 8 x i32> [[VEC_PHI5]])
+; CHECK-VS-NEXT:    [[TMP16]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> [[TMP15]], <vscale x 8 x i32> [[VEC_PHI5]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP13]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
+; CHECK-VS-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = call i32 @llvm.vector.reduce.umax.nxv8i32(<vscale x 8 x i32> [[TMP16]])
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP19]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i32 [[MAX_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %red = phi i32 [ 0, %entry ], [ %max, %loop ]
+  %gep = getelementptr inbounds i32, ptr %src, i64 %iv
+  %load = load i32, ptr %gep, align 1
+  %max = call i32 @llvm.umax(i32 %load, i32 %red)
+  %iv.next = add i64 %iv, 1
+  %icmp3 = icmp eq i64 %iv, %n
+  br i1 %icmp3, label %exit, label %loop
+
+exit:
+  ret i32 %max
+}
+
+define i64 @find_iv(ptr %src, i64 %n) {
+;
+; CHECK-LABEL: define i64 @find_iv(
+; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 16
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 15
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <16 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI1:%.*]] = phi <16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT:    [[TMP4]] = or <16 x i1> [[VEC_PHI1]], [[TMP3]]
+; CHECK-NEXT:    [[TMP5]] = select <16 x i1> [[TMP3]], <16 x i64> [[VEC_IND]], <16 x i64> [[VEC_PHI]]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add <16 x i64> [[VEC_IND]], splat (i64 16)
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = call i64 @llvm.vector.reduce.umax.v16i64(<16 x i64> [[TMP5]])
+; CHECK-NEXT:    [[TMP8:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP4]])
+; CHECK-NEXT:    [[TMP9:%.*]] = freeze i1 [[TMP8]]
+; CHECK-NEXT:    [[RDX_SELECT:%.*]] = select i1 [[TMP9]], i64 [[TMP7]], i64 0
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[RED:%.*]] = phi i64 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[SELECT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT:    [[LOAD:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-NEXT:    [[ICMP:%.*]] = icmp eq i8 [[LOAD]], 0
+; CHECK-NEXT:    [[SELECT]] = select i1 [[ICMP]], i64 [[IV]], i64 [[RED]]
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
+; CHECK-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[SELECT_LCSSA:%.*]] = phi i64 [ [[SELECT]], %[[LOOP]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i64 [[SELECT_LCSSA]]
+;
+; CHECK-VS-LABEL: define i64 @find_iv(
+; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ENTRY:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 4
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 4
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = call <vscale x 16 x i64> @llvm.stepvector.nxv16i64()
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i64> poison, i64 [[TMP3]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i64> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 16 x i64> [ [[TMP4]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i8>, ptr [[TMP5]], align 1
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = icmp eq <vscale x 16 x i8> [[WIDE_LOAD]], zeroinitializer
+; CHECK-VS-NEXT:    [[TMP7]] = or <vscale x 16 x i1> [[VEC_PHI1]], [[TMP6]]
+; CHECK-VS-NEXT:    [[TMP8]] = select <vscale x 16 x i1> [[TMP6]], <vscale x 16 x i64> [[VEC_IND]], <vscale x 16 x i64> [[VEC_PHI]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
+; CHECK-VS-NEXT:    [[VEC_IND_NEXT]] = add <vscale x 16 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = call i64 @llvm.vector.reduce.umax.nxv16i64(<vscale x 16 x i64> [[TMP8]])
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i1 @llvm.vector.reduce.or.nxv16i1(<vscale x 16 x i1> [[TMP7]])
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = freeze i1 [[TMP11]]
+; CHECK-VS-NEXT:    [[RDX_SELECT:%.*]] = select i1 [[TMP12]], i64 [[TMP10]], i64 0
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK-VS:       [[SCALAR_PH]]:
+; CHECK-VS-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
+; CHECK-VS:       [[LOOP]]:
+; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-VS-NEXT:    [[RED:%.*]] = phi i64 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[SELECT:%.*]], %[[LOOP]] ]
+; CHECK-VS-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-VS-NEXT:    [[LOAD:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-VS-NEXT:    [[ICMP:%.*]] = icmp eq i8 [[LOAD]], 0
+; CHECK-VS-NEXT:    [[SELECT]] = select i1 [[ICMP]], i64 [[IV]], i64 [[RED]]
+; CHECK-VS-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-VS-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
+; CHECK-VS-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    [[SELECT_LCSSA:%.*]] = phi i64 [ [[SELECT]], %[[LOOP]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i64 [[SELECT_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %red = phi i64 [ 0, %entry ], [ %select, %loop ]
+  %gep = getelementptr inbounds i8, ptr %src, i64 %iv
+  %load = load i8, ptr %gep, align 1
+  %icmp = icmp eq i8 %load, 0
+  %select = select i1 %icmp, i64 %iv, i64 %red
+  %iv.next = add i64 %iv, 1
+  %icmp3 = icmp eq i64 %iv, %n
+  br i1 %icmp3, label %exit, label %loop
+
+exit:
+  ret i64 %select
+}
+
+define i32 @any-of(ptr %src, i64 %n) {
+;
+; CHECK-LABEL: define i32 @any-of(
+; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP6:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP2]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP3]], align 1
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD3]], zeroinitializer
+; CHECK-NEXT:    [[TMP6]] = or <16 x i1> [[VEC_PHI]], [[TMP4]]
+; CHECK-NEXT:    [[TMP7]] = or <16 x i1> [[VEC_PHI2]], [[TMP5]]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[BIN_RDX:%.*]] = or <16 x i1> [[TMP7]], [[TMP6]]
+; CHECK-NEXT:    [[TMP9:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[BIN_RDX]])
+; CHECK-NEXT:    [[TMP10:%.*]] = freeze i1 [[TMP9]]
+; CHECK-NEXT:    [[RDX_SELECT:%.*]] = select i1 [[TMP10]], i32 1, i32 0
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[TMP11:%.*]] = icmp ne i32 [[BC_MERGE_RDX]], 0
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i1> poison, i1 [[TMP11]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i1> [[BROADCAST_SPLATINSERT]], <8 x i1> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i1> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP15:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP12:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP12]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
+; CHECK-NEXT:    [[TMP13:%.*]] = icmp eq <8 x i8> [[WIDE_MASKED_LOAD]], zeroinitializer
+; CHECK-NEXT:    [[TMP14:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP13]], <8 x i1> zeroinitializer
+; CHECK-NEXT:    [[TMP15]] = or <8 x i1> [[VEC_PHI5]], [[TMP14]]
+; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-NEXT:    [[TMP16:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP17:%.*]] = xor i1 [[TMP16]], true
+; CHECK-NEXT:    br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP18:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP15]])
+; CHECK-NEXT:    [[TMP19:%.*]] = freeze i1 [[TMP18]]
+; CHECK-NEXT:    [[RDX_SELECT7:%.*]] = select i1 [[TMP19]], i32 1, i32 0
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[SELECT_LCSSA:%.*]] = phi i32 [ [[RDX_SELECT7]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[SELECT_LCSSA]]
+;
+; CHECK-VS-LABEL: define i32 @any-of(
+; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP11:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i8, ptr [[TMP6]], i64 [[TMP4]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i8>, ptr [[TMP6]], align 1
+; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i8>, ptr [[TMP7]], align 1
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = icmp eq <vscale x 16 x i8> [[WIDE_LOAD]], zeroinitializer
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = icmp eq <vscale x 16 x i8> [[WIDE_LOAD3]], zeroinitializer
+; CHECK-VS-NEXT:    [[TMP10]] = or <vscale x 16 x i1> [[VEC_PHI]], [[TMP8]]
+; CHECK-VS-NEXT:    [[TMP11]] = or <vscale x 16 x i1> [[VEC_PHI2]], [[TMP9]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[BIN_RDX:%.*]] = or <vscale x 16 x i1> [[TMP11]], [[TMP10]]
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = call i1 @llvm.vector.reduce.or.nxv16i1(<vscale x 16 x i1> [[BIN_RDX]])
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = freeze i1 [[TMP13]]
+; CHECK-VS-NEXT:    [[RDX_SELECT:%.*]] = select i1 [[TMP14]], i32 1, i32 0
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = icmp ne i32 [[BC_MERGE_RDX]], 0
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = shl nuw i64 [[TMP16]], 3
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x i1> poison, i1 [[TMP15]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x i1> [[BROADCAST_SPLATINSERT]], <vscale x 8 x i1> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i1> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP21:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX4]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i8> @llvm.masked.load.nxv8i8.p0(ptr align 1 [[TMP18]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i8> poison)
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = icmp eq <vscale x 8 x i8> [[WIDE_MASKED_LOAD]], zeroinitializer
+; CHECK-VS-NEXT:    [[TMP20:%.*]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i1> [[TMP19]], <vscale x 8 x i1> zeroinitializer
+; CHECK-VS-NEXT:    [[TMP21]] = or <vscale x 8 x i1> [[VEC_PHI5]], [[TMP20]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP17]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
+; CHECK-VS-NEXT:    [[TMP22:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP23:%.*]] = xor i1 [[TMP22]], true
+; CHECK-VS-NEXT:    br i1 [[TMP23]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP24:%.*]] = call i1 @llvm.vector.reduce.or.nxv8i1(<vscale x 8 x i1> [[TMP21]])
+; CHECK-VS-NEXT:    [[TMP25:%.*]] = freeze i1 [[TMP24]]
+; CHECK-VS-NEXT:    [[RDX_SELECT7:%.*]] = select i1 [[TMP25]], i32 1, i32 0
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    [[SELECT_LCSSA:%.*]] = phi i32 [ [[RDX_SELECT7]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i32 [[SELECT_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %red = phi i32 [ 0, %entry ], [ %select, %loop ]
+  %gep = getelementptr inbounds i8, ptr %src, i64 %iv
+  %load = load i8, ptr %gep, align 1
+  %icmp = icmp eq i8 %load, 0
+  %select = select i1 %icmp, i32 1, i32 %red
+  %iv.next = add i64 %iv, 1
+  %icmp3 = icmp eq i64 %iv, %n
+  br i1 %icmp3, label %exit, label %loop
+
+exit:
+  ret i32 %select
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index ffde0d0084933..0518a67b21df8 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -132,294 +132,6 @@ exit:
   ret void
 }
 
-define i32 @add_redc(ptr %src, i64 %n) {
-; CHECK-LABEL: define i32 @add_redc(
-; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 16
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 1
-; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i32>, ptr [[TMP3]], align 1
-; CHECK-NEXT:    [[TMP4]] = add <16 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
-; CHECK-NEXT:    [[TMP5]] = add <16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[BIN_RDX:%.*]] = add <16 x i32> [[TMP5]], [[TMP4]]
-; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.add.v16i32(<16 x i32> [[BIN_RDX]])
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
-; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
-; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i32> [ [[TMP8]], %[[VEC_EPILOG_PH]] ], [ [[TMP11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
-; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[TMP9]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
-; CHECK-NEXT:    [[TMP10:%.*]] = add <8 x i32> [[WIDE_MASKED_LOAD]], [[VEC_PHI5]]
-; CHECK-NEXT:    [[TMP11]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP10]], <8 x i32> [[VEC_PHI5]]
-; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-NEXT:    [[TMP13:%.*]] = xor i1 [[TMP12]], true
-; CHECK-NEXT:    br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP11]])
-; CHECK-NEXT:    br label %[[EXIT]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP14]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
-; CHECK-NEXT:    ret i32 [[ADD_LCSSA]]
-;
-; CHECK-VS-LABEL: define i32 @add_redc(
-; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
-; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
-; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-VS:       [[VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP4]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 1
-; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
-; CHECK-VS-NEXT:    [[TMP8]] = add <vscale x 16 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
-; CHECK-VS-NEXT:    [[TMP9]] = add <vscale x 16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-VS:       [[MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[BIN_RDX:%.*]] = add <vscale x 16 x i32> [[TMP9]], [[TMP8]]
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i32 @llvm.vector.reduce.add.nxv16i32(<vscale x 16 x i32> [[BIN_RDX]])
-; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
-; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = insertelement <vscale x 8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i32> [ [[TMP14]], %[[VEC_EPILOG_PH]] ], [ [[TMP17:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
-; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 1 [[TMP15]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
-; CHECK-VS-NEXT:    [[TMP16:%.*]] = add <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], [[VEC_PHI5]]
-; CHECK-VS-NEXT:    [[TMP17]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> [[TMP16]], <vscale x 8 x i32> [[VEC_PHI5]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP13]]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
-; CHECK-VS-NEXT:    [[TMP18:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP19:%.*]] = xor i1 [[TMP18]], true
-; CHECK-VS-NEXT:    br i1 [[TMP19]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP20:%.*]] = call i32 @llvm.vector.reduce.add.nxv8i32(<vscale x 8 x i32> [[TMP17]])
-; CHECK-VS-NEXT:    br label %[[EXIT]]
-; CHECK-VS:       [[EXIT]]:
-; CHECK-VS-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP20]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
-; CHECK-VS-NEXT:    ret i32 [[ADD_LCSSA]]
-;
-entry:
-  br label %loop
-
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %red = phi i32 [ 0, %entry ], [ %add, %loop ]
-  %gep = getelementptr inbounds i32, ptr %src, i64 %iv
-  %load = load i32, ptr %gep, align 1
-  %add = add i32 %load, %red
-  %iv.next = add i64 %iv, 1
-  %icmp3 = icmp eq i64 %iv, %n
-  br i1 %icmp3, label %exit, label %loop
-
-exit:
-  ret i32 %add
-}
-
-define i32 @max_redc(ptr %src, i64 %n) {
-; CHECK-LABEL: define i32 @max_redc(
-; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 16
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 1
-; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i32>, ptr [[TMP3]], align 1
-; CHECK-NEXT:    [[TMP4]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD]], <16 x i32> [[VEC_PHI]])
-; CHECK-NEXT:    [[TMP5]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD3]], <16 x i32> [[VEC_PHI2]])
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[RDX_MINMAX:%.*]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[TMP4]], <16 x i32> [[TMP5]])
-; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.umax.v16i32(<16 x i32> [[RDX_MINMAX]])
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
-; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i32> poison, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT]], <8 x i32> poison, <8 x i32> zeroinitializer
-; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i32> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
-; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[TMP8]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
-; CHECK-NEXT:    [[TMP9:%.*]] = call <8 x i32> @llvm.umax.v8i32(<8 x i32> [[WIDE_MASKED_LOAD]], <8 x i32> [[VEC_PHI5]])
-; CHECK-NEXT:    [[TMP10]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP9]], <8 x i32> [[VEC_PHI5]]
-; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-NEXT:    [[TMP12:%.*]] = xor i1 [[TMP11]], true
-; CHECK-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[TMP13:%.*]] = call i32 @llvm.vector.reduce.umax.v8i32(<8 x i32> [[TMP10]])
-; CHECK-NEXT:    br label %[[EXIT]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP13]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
-; CHECK-NEXT:    ret i32 [[MAX_LCSSA]]
-;
-; CHECK-VS-LABEL: define i32 @max_redc(
-; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
-; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
-; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-VS:       [[VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP4]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 1
-; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
-; CHECK-VS-NEXT:    [[TMP8]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD]], <vscale x 16 x i32> [[VEC_PHI]])
-; CHECK-VS-NEXT:    [[TMP9]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], <vscale x 16 x i32> [[VEC_PHI2]])
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-VS:       [[MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[RDX_MINMAX:%.*]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[TMP8]], <vscale x 16 x i32> [[TMP9]])
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i32 @llvm.vector.reduce.umax.nxv16i32(<vscale x 16 x i32> [[RDX_MINMAX]])
-; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
-; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i32> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP16:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
-; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 1 [[TMP14]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = call <vscale x 8 x i32> @llvm.umax.nxv8i32(<vscale x 8 x i32> [[WIDE_MASKED_LOAD]], <vscale x 8 x i32> [[VEC_PHI5]])
-; CHECK-VS-NEXT:    [[TMP16]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> [[TMP15]], <vscale x 8 x i32> [[VEC_PHI5]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP13]]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
-; CHECK-VS-NEXT:    [[TMP17:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
-; CHECK-VS-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP19:%.*]] = call i32 @llvm.vector.reduce.umax.nxv8i32(<vscale x 8 x i32> [[TMP16]])
-; CHECK-VS-NEXT:    br label %[[EXIT]]
-; CHECK-VS:       [[EXIT]]:
-; CHECK-VS-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP19]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
-; CHECK-VS-NEXT:    ret i32 [[MAX_LCSSA]]
-;
-entry:
-  br label %loop
-
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %red = phi i32 [ 0, %entry ], [ %max, %loop ]
-  %gep = getelementptr inbounds i32, ptr %src, i64 %iv
-  %load = load i32, ptr %gep, align 1
-  %max = call i32 @llvm.umax(i32 %load, i32 %red)
-  %iv.next = add i64 %iv, 1
-  %icmp3 = icmp eq i64 %iv, %n
-  br i1 %icmp3, label %exit, label %loop
-
-exit:
-  ret i32 %max
-}
-
 define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-LABEL: define i32 @live-out(
 ; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {

>From 5158b3592165faa1c4c9a1ec390aee590a9e3a63 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Fri, 4 Sep 2026 18:14:59 +0100
Subject: [PATCH 13/25] rebase + fix crash of 1st recurrence + fix crash when
 start value of WIDEN-CANONICAL-INDUCTION is not CANONICAL-IV

---
 .../Transforms/Vectorize/LoopVectorize.cpp    |  18 +-
 .../Transforms/Vectorize/VPlanLowering.cpp    |  34 +-
 .../AArch64/fold-epilogue-tail-reductions.ll  | 258 +++++++-
 .../AArch64/fold-epilogue-tail.ll             | 286 ++++++---
 .../AArch64/partial-reduce-with-predicate.ll  | 563 ++++++++++++++++++
 ...g-vectorization-fixed-order-recurrences.ll | 173 +++++-
 .../LoopVectorize/fold-epilogue-tail.ll       |  38 +-
 7 files changed, 1237 insertions(+), 133 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 7f6a3c0b6b2c1..12dcca30044bd 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3169,7 +3169,7 @@ void LoopVectorizationPlanner::emitInvalidCostRemarks(
   for (const auto &Plan : VPlans) {
     // Skip cost remarks when Plan is not compatible with the CM.
     // Specifically for the case of epilogue tail-folded Plans.
-    if (Plan->hasTailFolded() ^ CM->preferTailFoldedLoop())
+    if (Plan->hasTailFolded() != CM->preferTailFoldedLoop())
       continue;
     for (ElementCount VF : Plan->vectorFactors()) {
       // The VPlan-based cost model is designed for computing vector cost.
@@ -5419,15 +5419,13 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 
       // For scalar VF, skip VPlan cost check as VPlan cost is designed for
       // vector VFs only.
-      if (!VPlans.empty() &&
+      if (!VPlans.empty() && (VPlans.front()->getSingleVF() == UserVF) &&
           (UserVF.isScalar() ||
            cost(*VPlans.front(), UserVF, /*RU=*/nullptr, *CM).isValid())) {
         // Plan for epilogue only if we succeeded in building main loop vplan.
-
         // Try to plan for tail-folded epilogue if it's enabled/doable,
         // otherwise plan for unpredicated epilogue:
-        bool EpilogueTfPlanCreated = planForEpilogueTF();
-        if (!EpilogueTfPlanCreated) {
+        if (!planForEpilogueTF()) {
           ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
           if (EpilogueUserVF.isVector() &&
               ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
@@ -5496,6 +5494,7 @@ bool LoopVectorizationPlanner::planForEpilogueTF() {
     reportVectorizationInfo(
         "Failed to build initial tail-folded epilogue VPlan",
         "InvalidTailFoldedEpilogue", ORE, OrigLoop);
+    assert(false && "Failed to build initial tail-folded epilogue VPlan");
     return false;
   }
 
@@ -7541,7 +7540,7 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
   auto *Increment = vputils::findCanonicalIVIncrement(Plan);
   assert(Increment && "Must have a canonical IV increment at this point");
   IV->replaceUsesWithIf(Add, [Add, Increment](VPUser &U, unsigned) {
-    return &U != Add && &U != Increment && !isa<VPWidenCanonicalIVRecipe>(&U);
+    return &U != Add && &U != Increment;
   });
   VPInstruction *OffsetIVInc =
       VPBuilder::getToInsertAfter(Increment).createAdd(Increment, VPV);
@@ -7664,11 +7663,16 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
           VPInstruction::CanonicalIVIncrementForPart, {VPV, &Plan.getVF()}, {},
           R.getDebugLoc(), "index.part.next");
       auto *EntryALM = EntryBuilder.createNaryOp(
-          VPInstruction::ActiveLaneMask,
+          VPInstruction::WideActiveLaneMask,
           {EntryIncrement, Plan.getTripCount(), ALMMultiplier}, R.getDebugLoc(),
           "active.lane.mask.entry");
       cast<VPHeaderPHIRecipe>(&R)->setStartValue(EntryALM);
       continue;
+    } else if (isa<VPFirstOrderRecurrencePHIRecipe>(&R)) {
+      auto *RecPhi = cast<VPFirstOrderRecurrencePHIRecipe>(&R);
+      VPInstruction *ResumeForEpi =
+          IRPhiToResumeForEpi.at(cast<PHINode>(RecPhi->getUnderlyingInstr()));
+      ResumeV = ResumeForEpi->getUnderlyingValue();
     } else {
       // Retrieve the induction resume value via ResumeForEpilogue.
       PHINode *IndPhi = cast<VPWidenInductionRecipe>(&R)->getPHINode();
diff --git a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
index 230fb5f39d33f..5a9124b8eea3a 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
@@ -45,12 +45,36 @@ void VPlanTransforms::replaceWideCanonicalIVWithWideIV(
   if (!LoopRegion)
     return;
 
-  auto *WideCanIV =
-      findUserOf<VPWidenCanonicalIVRecipe>(LoopRegion->getCanonicalIV());
+  VPWidenCanonicalIVRecipe *WideCanIV = nullptr;
+  VPIRValue *StartValue = nullptr;
+  // VPWidenCanonicalIVRecipe is either a direct user of CanonicalIV or
+  // Add (CanonicalIV, resumeValue) (like the case for tail-folded epilogue).
+  for (auto *User : LoopRegion->getCanonicalIV()->users()) {
+    if (isa<VPWidenCanonicalIVRecipe>(User)) {
+      WideCanIV = cast<VPWidenCanonicalIVRecipe>(User);
+      StartValue = Plan.getZero(WideCanIV->getScalarType());
+      break;
+    }
+    if (isa<VPInstruction>(User)) {
+      auto *UserInstr = cast<VPInstruction>(User);
+      auto *It = find_if(UserInstr->users(), IsaPred<VPWidenCanonicalIVRecipe>);
+      if (It != UserInstr->user_end()) {
+        WideCanIV = cast<VPWidenCanonicalIVRecipe>(*It);
+        match(UserInstr, m_Add(m_VPValue(), m_VPIRValue(StartValue)));
+        assert(StartValue &&
+               "WIDEN-CANONICAL-INDUCTION is only expected to be reached "
+               "through the canonical IV directly, or through a single 'add "
+               "CanonicalIV, StartValue' introduced for epilogue-loop resume "
+               "values; found a different pattern here");
+        if (!StartValue)
+          return;
+      }
+    }
+  }
   if (!WideCanIV)
     return;
 
-  Type *CanIVTy = LoopRegion->getCanonicalIVType();
+  Type *CanIVTy = WideCanIV->getScalarType();
 
   // Replace the wide canonical IV with a scalar-iv-steps over the canonical
   // IV.
@@ -58,7 +82,7 @@ void VPlanTransforms::replaceWideCanonicalIVWithWideIV(
     VPBuilder Builder(WideCanIV);
     WideCanIV->replaceAllUsesWith(vputils::createScalarIVSteps(
         Plan, InductionDescriptor::IK_IntInduction, Instruction::Add, nullptr,
-        nullptr, Plan.getZero(CanIVTy), Plan.getConstantInt(CanIVTy, 1),
+        nullptr, StartValue, Plan.getConstantInt(CanIVTy, 1),
         WideCanIV->getDebugLoc(), Builder,
         {static_cast<bool>(WideCanIV->getNoWrapFlags().HasNUW), false}));
     WideCanIV->eraseFromParent();
@@ -108,7 +132,7 @@ void VPlanTransforms::replaceWideCanonicalIVWithWideIV(
       InductionDescriptor::getCanonicalIntInduction(CanIVTy, SE);
   VPValue *StepV = Plan.getConstantInt(CanIVTy, 1);
   auto *NewWideIV = new VPWidenIntOrFpInductionRecipe(
-      /*IV=*/nullptr, Plan.getZero(CanIVTy), StepV, &Plan.getVF(), ID,
+      /*IV=*/nullptr, StartValue, StepV, &Plan.getVF(), ID,
       WideCanIV->getNoWrapFlags(), WideCanIV->getDebugLoc());
   NewWideIV->insertBefore(&*Header->getFirstNonPhi());
   WideCanIV->replaceAllUsesWith(NewWideIV);
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll
index 1b2088b617606..1c9c05884b7d4 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll
@@ -82,8 +82,7 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
 ; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
 ; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
 ; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VECTOR_BODY]]:
@@ -96,7 +95,7 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
 ; CHECK-VS-NEXT:    [[TMP8]] = add <vscale x 16 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
 ; CHECK-VS-NEXT:    [[TMP9]] = add <vscale x 16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
 ; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-VS:       [[MIDDLE_BLOCK]]:
@@ -226,8 +225,7 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
 ; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
 ; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
 ; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VECTOR_BODY]]:
@@ -240,7 +238,7 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
 ; CHECK-VS-NEXT:    [[TMP8]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD]], <vscale x 16 x i32> [[VEC_PHI]])
 ; CHECK-VS-NEXT:    [[TMP9]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], <vscale x 16 x i32> [[VEC_PHI2]])
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
 ; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-VS:       [[MIDDLE_BLOCK]]:
@@ -356,11 +354,10 @@ define i64 @find_iv(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
 ; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 4
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
 ; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
 ; CHECK-VS-NEXT:    [[TMP4:%.*]] = call <vscale x 16 x i64> @llvm.stepvector.nxv16i64()
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i64> poison, i64 [[TMP3]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i64> poison, i64 [[TMP2]], i64 0
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i64> poison, <vscale x 16 x i32> zeroinitializer
 ; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VECTOR_BODY]]:
@@ -373,7 +370,7 @@ define i64 @find_iv(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[TMP6:%.*]] = icmp eq <vscale x 16 x i8> [[WIDE_LOAD]], zeroinitializer
 ; CHECK-VS-NEXT:    [[TMP7]] = or <vscale x 16 x i1> [[VEC_PHI1]], [[TMP6]]
 ; CHECK-VS-NEXT:    [[TMP8]] = select <vscale x 16 x i1> [[TMP6]], <vscale x 16 x i64> [[VEC_IND]], <vscale x 16 x i64> [[VEC_PHI]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
 ; CHECK-VS-NEXT:    [[VEC_IND_NEXT]] = add <vscale x 16 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
 ; CHECK-VS-NEXT:    [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
@@ -504,8 +501,7 @@ define i32 @any-of(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
 ; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = shl nuw i64 [[TMP4]], 1
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP5]]
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
 ; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
 ; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VECTOR_BODY]]:
@@ -520,7 +516,7 @@ define i32 @any-of(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[TMP9:%.*]] = icmp eq <vscale x 16 x i8> [[WIDE_LOAD3]], zeroinitializer
 ; CHECK-VS-NEXT:    [[TMP10]] = or <vscale x 16 x i1> [[VEC_PHI]], [[TMP8]]
 ; CHECK-VS-NEXT:    [[TMP11]] = or <vscale x 16 x i1> [[VEC_PHI2]], [[TMP9]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP5]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
 ; CHECK-VS-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-VS:       [[MIDDLE_BLOCK]]:
@@ -582,3 +578,239 @@ loop:
 exit:
   ret i32 %select
 }
+
+define i64 @arg_min_first_index(ptr %arr, i64 %n, i64 %start) {
+; CHECK-LABEL: define i64 @arg_min_first_index(
+; CHECK-SAME: ptr [[ARR:%.*]], i64 [[N:%.*]], i64 [[START:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 16
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 15
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <16 x i64> poison, i64 [[START]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <16 x i64> [[BROADCAST_SPLATINSERT]], <16 x i64> poison, <16 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <16 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i64> [ [[BROADCAST_SPLAT]], %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i64> [ poison, %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i64>, ptr [[TMP1]], align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = icmp slt <16 x i64> [[WIDE_LOAD]], [[VEC_PHI]]
+; CHECK-NEXT:    [[TMP3]] = select <16 x i1> [[TMP2]], <16 x i64> [[VEC_IND]], <16 x i64> [[VEC_PHI2]]
+; CHECK-NEXT:    [[TMP4]] = call <16 x i64> @llvm.smin.v16i64(<16 x i64> [[WIDE_LOAD]], <16 x i64> [[VEC_PHI]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <16 x i64> [[VEC_IND]], splat (i64 16)
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP6:%.*]] = call i64 @llvm.vector.reduce.smin.v16i64(<16 x i64> [[TMP4]])
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <16 x i64> poison, i64 [[TMP6]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <16 x i64> [[BROADCAST_SPLATINSERT3]], <16 x i64> poison, <16 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq <16 x i64> [[TMP4]], [[BROADCAST_SPLAT4]]
+; CHECK-NEXT:    [[TMP8:%.*]] = select <16 x i1> [[TMP7]], <16 x i64> [[TMP3]], <16 x i64> splat (i64 -1)
+; CHECK-NEXT:    [[TMP9:%.*]] = call i64 @llvm.vector.reduce.umin.v16i64(<16 x i64> [[TMP8]])
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[TMP6]], [[START]]
+; CHECK-NEXT:    [[TMP11:%.*]] = select i1 [[TMP10]], i64 0, i64 [[TMP9]]
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
+; CHECK-NEXT:    br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF11:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[TMP6]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[START]], %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX5:%.*]] = phi i64 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[TMP12:%.*]] = and i64 [[N]], 7
+; CHECK-NEXT:    [[N_VEC6:%.*]] = sub i64 [[N]], [[TMP12]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT7:%.*]] = insertelement <8 x i64> poison, i64 [[BC_MERGE_RDX]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT8:%.*]] = shufflevector <8 x i64> [[BROADCAST_SPLATINSERT7]], <8 x i64> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT9:%.*]] = insertelement <8 x i64> poison, i64 [[BC_MERGE_RDX5]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT10:%.*]] = shufflevector <8 x i64> [[BROADCAST_SPLATINSERT9]], <8 x i64> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT11:%.*]] = insertelement <8 x i64> poison, i64 [[VEC_EPILOG_RESUME_VAL]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT12:%.*]] = shufflevector <8 x i64> [[BROADCAST_SPLATINSERT11]], <8 x i64> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[INDUCTION:%.*]] = add nuw nsw <8 x i64> [[BROADCAST_SPLAT12]], <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7>
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX13:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT18:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_IND14:%.*]] = phi <8 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT19:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI15:%.*]] = phi <8 x i64> [ [[BROADCAST_SPLAT8]], %[[VEC_EPILOG_PH]] ], [ [[TMP16:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI16:%.*]] = phi <8 x i64> [ [[BROADCAST_SPLAT10]], %[[VEC_EPILOG_PH]] ], [ [[TMP15:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP13:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[INDEX13]]
+; CHECK-NEXT:    [[WIDE_LOAD17:%.*]] = load <8 x i64>, ptr [[TMP13]], align 8
+; CHECK-NEXT:    [[TMP14:%.*]] = icmp slt <8 x i64> [[WIDE_LOAD17]], [[VEC_PHI15]]
+; CHECK-NEXT:    [[TMP15]] = select <8 x i1> [[TMP14]], <8 x i64> [[VEC_IND14]], <8 x i64> [[VEC_PHI16]]
+; CHECK-NEXT:    [[TMP16]] = call <8 x i64> @llvm.smin.v8i64(<8 x i64> [[WIDE_LOAD17]], <8 x i64> [[VEC_PHI15]])
+; CHECK-NEXT:    [[INDEX_NEXT18]] = add nuw i64 [[INDEX13]], 8
+; CHECK-NEXT:    [[VEC_IND_NEXT19]] = add nuw nsw <8 x i64> [[VEC_IND14]], splat (i64 8)
+; CHECK-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT18]], [[N_VEC6]]
+; CHECK-NEXT:    br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP18:%.*]] = call i64 @llvm.vector.reduce.smin.v8i64(<8 x i64> [[TMP16]])
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT20:%.*]] = insertelement <8 x i64> poison, i64 [[TMP18]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT21:%.*]] = shufflevector <8 x i64> [[BROADCAST_SPLATINSERT20]], <8 x i64> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP19:%.*]] = icmp eq <8 x i64> [[TMP16]], [[BROADCAST_SPLAT21]]
+; CHECK-NEXT:    [[TMP20:%.*]] = select <8 x i1> [[TMP19]], <8 x i64> [[TMP15]], <8 x i64> splat (i64 -1)
+; CHECK-NEXT:    [[TMP21:%.*]] = call i64 @llvm.vector.reduce.umin.v8i64(<8 x i64> [[TMP20]])
+; CHECK-NEXT:    [[TMP22:%.*]] = icmp eq i64 [[TMP18]], [[START]]
+; CHECK-NEXT:    [[TMP23:%.*]] = select i1 [[TMP22]], i64 0, i64 [[TMP21]]
+; CHECK-NEXT:    [[CMP_N22:%.*]] = icmp eq i64 [[N]], [[N_VEC6]]
+; CHECK-NEXT:    br i1 [[CMP_N22]], label %[[FOR_END]], label %[[VEC_EPILOG_SCALAR_PH]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC6]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX23:%.*]] = phi i64 [ [[TMP18]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP6]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[START]], %[[ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX24:%.*]] = phi i64 [ [[TMP23]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[MIN:%.*]] = phi i64 [ [[BC_MERGE_RDX23]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MIN_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[MIN_LOC:%.*]] = phi i64 [ [[BC_MERGE_RDX24]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MIN_LOC_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[IV]]
+; CHECK-NEXT:    [[TMP24:%.*]] = load i64, ptr [[ARRAYIDX]], align 8
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[TMP24]], [[MIN]]
+; CHECK-NEXT:    [[MIN_LOC_NEXT]] = select i1 [[CMP]], i64 [[IV]], i64 [[MIN_LOC]]
+; CHECK-NEXT:    [[MIN_NEXT]] = tail call i64 @llvm.smin.i64(i64 [[TMP24]], i64 [[MIN]])
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[EXITCOND_NOT]], label %[[FOR_END]], label %[[FOR_BODY]]
+; CHECK:       [[FOR_END]]:
+; CHECK-NEXT:    [[MIN_LOC_NEXT_LCSSA:%.*]] = phi i64 [ [[MIN_LOC_NEXT]], %[[FOR_BODY]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ], [ [[TMP23]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i64 [[MIN_LOC_NEXT_LCSSA]]
+;
+; CHECK-VS-LABEL: define i64 @arg_min_first_index(
+; CHECK-VS-SAME: ptr [[ARR:%.*]], i64 [[N:%.*]], i64 [[START:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i64> poison, i64 [[START]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i64> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = call <vscale x 16 x i64> @llvm.stepvector.nxv16i64()
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <vscale x 16 x i64> poison, i64 [[TMP2]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <vscale x 16 x i64> [[BROADCAST_SPLATINSERT2]], <vscale x 16 x i64> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 16 x i64> [ [[TMP3]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i64> [ [[BROADCAST_SPLAT]], %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI4:%.*]] = phi <vscale x 16 x i64> [ poison, %[[VECTOR_PH]] ], [ [[TMP6:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i64>, ptr [[TMP4]], align 8
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = icmp slt <vscale x 16 x i64> [[WIDE_LOAD]], [[VEC_PHI]]
+; CHECK-VS-NEXT:    [[TMP6]] = select <vscale x 16 x i1> [[TMP5]], <vscale x 16 x i64> [[VEC_IND]], <vscale x 16 x i64> [[VEC_PHI4]]
+; CHECK-VS-NEXT:    [[TMP7]] = call <vscale x 16 x i64> @llvm.smin.nxv16i64(<vscale x 16 x i64> [[WIDE_LOAD]], <vscale x 16 x i64> [[VEC_PHI]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-VS-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <vscale x 16 x i64> [[VEC_IND]], [[BROADCAST_SPLAT3]]
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = call i64 @llvm.vector.reduce.smin.nxv16i64(<vscale x 16 x i64> [[TMP7]])
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT5:%.*]] = insertelement <vscale x 16 x i64> poison, i64 [[TMP9]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT6:%.*]] = shufflevector <vscale x 16 x i64> [[BROADCAST_SPLATINSERT5]], <vscale x 16 x i64> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq <vscale x 16 x i64> [[TMP7]], [[BROADCAST_SPLAT6]]
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = select <vscale x 16 x i1> [[TMP10]], <vscale x 16 x i64> [[TMP6]], <vscale x 16 x i64> splat (i64 -1)
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vector.reduce.umin.nxv16i64(<vscale x 16 x i64> [[TMP11]])
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[TMP9]], [[START]]
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = select i1 [[TMP13]], i64 0, i64 [[TMP12]]
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], [[TMP1]]
+; CHECK-VS-NEXT:    br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF11:![0-9]+]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[TMP9]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[START]], %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX7:%.*]] = phi i64 [ [[TMP14]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = shl nuw i64 [[TMP15]], 3
+; CHECK-VS-NEXT:    [[N_MOD_VF8:%.*]] = urem i64 [[N]], [[TMP16]]
+; CHECK-VS-NEXT:    [[N_VEC9:%.*]] = sub i64 [[N]], [[N_MOD_VF8]]
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT10:%.*]] = insertelement <vscale x 8 x i64> poison, i64 [[BC_MERGE_RDX]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT11:%.*]] = shufflevector <vscale x 8 x i64> [[BROADCAST_SPLATINSERT10]], <vscale x 8 x i64> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT12:%.*]] = insertelement <vscale x 8 x i64> poison, i64 [[BC_MERGE_RDX7]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT13:%.*]] = shufflevector <vscale x 8 x i64> [[BROADCAST_SPLATINSERT12]], <vscale x 8 x i64> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.stepvector.nxv8i64()
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT14:%.*]] = insertelement <vscale x 8 x i64> poison, i64 [[VEC_EPILOG_RESUME_VAL]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT15:%.*]] = shufflevector <vscale x 8 x i64> [[BROADCAST_SPLATINSERT14]], <vscale x 8 x i64> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[INDUCTION:%.*]] = add nuw nsw <vscale x 8 x i64> [[BROADCAST_SPLAT15]], [[TMP17]]
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT16:%.*]] = insertelement <vscale x 8 x i64> poison, i64 [[TMP16]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT17:%.*]] = shufflevector <vscale x 8 x i64> [[BROADCAST_SPLATINSERT16]], <vscale x 8 x i64> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX18:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT23:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_IND19:%.*]] = phi <vscale x 8 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT24:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI20:%.*]] = phi <vscale x 8 x i64> [ [[BROADCAST_SPLAT11]], %[[VEC_EPILOG_PH]] ], [ [[TMP21:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VEC_PHI21:%.*]] = phi <vscale x 8 x i64> [ [[BROADCAST_SPLAT13]], %[[VEC_EPILOG_PH]] ], [ [[TMP20:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[INDEX18]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD22:%.*]] = load <vscale x 8 x i64>, ptr [[TMP18]], align 8
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = icmp slt <vscale x 8 x i64> [[WIDE_LOAD22]], [[VEC_PHI20]]
+; CHECK-VS-NEXT:    [[TMP20]] = select <vscale x 8 x i1> [[TMP19]], <vscale x 8 x i64> [[VEC_IND19]], <vscale x 8 x i64> [[VEC_PHI21]]
+; CHECK-VS-NEXT:    [[TMP21]] = call <vscale x 8 x i64> @llvm.smin.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD22]], <vscale x 8 x i64> [[VEC_PHI20]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT23]] = add nuw i64 [[INDEX18]], [[TMP16]]
+; CHECK-VS-NEXT:    [[VEC_IND_NEXT24]] = add nuw nsw <vscale x 8 x i64> [[VEC_IND19]], [[BROADCAST_SPLAT17]]
+; CHECK-VS-NEXT:    [[TMP22:%.*]] = icmp eq i64 [[INDEX_NEXT23]], [[N_VEC9]]
+; CHECK-VS-NEXT:    br i1 [[TMP22]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP23:%.*]] = call i64 @llvm.vector.reduce.smin.nxv8i64(<vscale x 8 x i64> [[TMP21]])
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT25:%.*]] = insertelement <vscale x 8 x i64> poison, i64 [[TMP23]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT26:%.*]] = shufflevector <vscale x 8 x i64> [[BROADCAST_SPLATINSERT25]], <vscale x 8 x i64> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[TMP24:%.*]] = icmp eq <vscale x 8 x i64> [[TMP21]], [[BROADCAST_SPLAT26]]
+; CHECK-VS-NEXT:    [[TMP25:%.*]] = select <vscale x 8 x i1> [[TMP24]], <vscale x 8 x i64> [[TMP20]], <vscale x 8 x i64> splat (i64 -1)
+; CHECK-VS-NEXT:    [[TMP26:%.*]] = call i64 @llvm.vector.reduce.umin.nxv8i64(<vscale x 8 x i64> [[TMP25]])
+; CHECK-VS-NEXT:    [[TMP27:%.*]] = icmp eq i64 [[TMP23]], [[START]]
+; CHECK-VS-NEXT:    [[TMP28:%.*]] = select i1 [[TMP27]], i64 0, i64 [[TMP26]]
+; CHECK-VS-NEXT:    [[CMP_N27:%.*]] = icmp eq i64 [[N]], [[N_VEC9]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N27]], label %[[FOR_END]], label %[[VEC_EPILOG_SCALAR_PH]]
+; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-VS-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX28:%.*]] = phi i64 [ [[TMP23]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP9]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[START]], %[[ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[BC_MERGE_RDX29:%.*]] = phi i64 [ [[TMP28]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP14]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-VS-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-VS:       [[FOR_BODY]]:
+; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-VS-NEXT:    [[MIN:%.*]] = phi i64 [ [[BC_MERGE_RDX28]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MIN_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-VS-NEXT:    [[MIN_LOC:%.*]] = phi i64 [ [[BC_MERGE_RDX29]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MIN_LOC_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[IV]]
+; CHECK-VS-NEXT:    [[TMP29:%.*]] = load i64, ptr [[ARRAYIDX]], align 8
+; CHECK-VS-NEXT:    [[CMP:%.*]] = icmp slt i64 [[TMP29]], [[MIN]]
+; CHECK-VS-NEXT:    [[MIN_LOC_NEXT]] = select i1 [[CMP]], i64 [[IV]], i64 [[MIN_LOC]]
+; CHECK-VS-NEXT:    [[MIN_NEXT]] = tail call i64 @llvm.smin.i64(i64 [[TMP29]], i64 [[MIN]])
+; CHECK-VS-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-VS-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-VS-NEXT:    br i1 [[EXITCOND_NOT]], label %[[FOR_END]], label %[[FOR_BODY]]
+; CHECK-VS:       [[FOR_END]]:
+; CHECK-VS-NEXT:    [[MIN_LOC_NEXT_LCSSA:%.*]] = phi i64 [ [[MIN_LOC_NEXT]], %[[FOR_BODY]] ], [ [[TMP14]], %[[MIDDLE_BLOCK]] ], [ [[TMP28]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i64 [[MIN_LOC_NEXT_LCSSA]]
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+  %min = phi i64 [ %start, %entry ], [ %min.next, %for.body ]
+  %min.loc = phi i64 [ 0, %entry ], [ %min.loc.next, %for.body ]
+  %arrayidx = getelementptr inbounds nuw [8 x i8], ptr %arr, i64 %iv
+  %0 = load i64, ptr %arrayidx
+  %cmp = icmp slt i64 %0, %min
+  %min.loc.next = select i1 %cmp, i64 %iv, i64 %min.loc
+  %min.next = tail call i64 @llvm.smin.i64(i64 %0, i64 %min)
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond.not = icmp eq i64 %iv.next, %n
+  br i1 %exitcond.not, label %for.end, label %for.body
+
+for.end:
+  ret i64 %min.loc.next
+}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 0518a67b21df8..3cee861a822b4 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,12 +1,11 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; REQUIRES: asserts
-; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -debug-only=loop-vectorize -mattr=+sve -S %s | FileCheck %s
 
-; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width="vscale x 16" -epilogue-vectorization-force-VF="vscale x 8" -debug-only=loop-vectorize -mattr=+sve -S %s | FileCheck %s --check-prefix=CHECK-VS
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width=16 -epilogue-vectorization-force-VF=8 -mattr=+sve %s | FileCheck %s
 
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize,vectorutils --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width="vscale x 16" -epilogue-vectorization-force-VF="vscale x 8" -mattr=+sve %s | FileCheck %s --check-prefix=CHECK-VS
+
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail -debug-only=loop-vectorize,vectorutils --disable-output \
 ; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALIDATE-INTERLEAVE
 
 target triple = "aarch64-linux-gnu"
@@ -74,21 +73,20 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
 ; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 1
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP4]]
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
 ; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[VAL]], i64 0
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
 ; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VECTOR_BODY]]:
 ; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[TMP5]], i64 [[TMP3]]
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 [[TMP3]]
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP4]], align 4
 ; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP5]], align 4
-; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP6]], align 4
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-VS:       [[MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
@@ -96,8 +94,8 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP8:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP9:%.*]] = shl nuw i64 [[TMP8]], 3
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = shl nuw i64 [[TMP7]], 3
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[VAL]], i64 0
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT2]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
@@ -105,13 +103,13 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
 ; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT5:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX4]]
-; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[BROADCAST_SPLAT3]], ptr align 4 [[TMP10]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-VS-NEXT:    [[INDEX_NEXT5]] = add i64 [[INDEX4]], [[TMP9]]
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX4]]
+; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[BROADCAST_SPLAT3]], ptr align 4 [[TMP9]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT5]] = add i64 [[INDEX4]], [[TMP8]]
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT5]], i64 [[N]])
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = xor i1 [[TMP11]], true
-; CHECK-VS-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = xor i1 [[TMP10]], true
+; CHECK-VS-NEXT:    br i1 [[TMP11]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    br label %[[EXIT]]
 ; CHECK-VS:       [[EXIT]]:
@@ -196,51 +194,50 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
 ; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP3]], 1
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP4]]
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
 ; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
 ; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VECTOR_BODY]]:
 ; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP5]], i64 [[TMP3]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 4
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP4]]
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP4]], i64 [[TMP3]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP5]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-VS:       [[MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP8:%.*]] = call i32 @llvm.vscale.i32()
-; CHECK-VS-NEXT:    [[TMP9:%.*]] = mul nuw i32 [[TMP8]], 16
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = sub i32 [[TMP9]], 1
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = extractelement <vscale x 16 x i32> [[WIDE_LOAD]], i32 [[TMP10]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = mul nuw i32 [[TMP7]], 16
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = sub i32 [[TMP8]], 1
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = extractelement <vscale x 16 x i32> [[WIDE_LOAD]], i32 [[TMP9]]
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
 ; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = shl nuw i64 [[TMP11]], 3
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
 ; CHECK-VS-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX2]]
-; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 4 [[TMP14]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
-; CHECK-VS-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], [[TMP13]]
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX2]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 4 [[TMP13]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], [[TMP12]]
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP16:%.*]] = xor i1 [[TMP15]], true
-; CHECK-VS-NEXT:    br i1 [[TMP16]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = xor i1 [[TMP14]], true
+; CHECK-VS-NEXT:    br i1 [[TMP15]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP17:%.*]] = xor <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
-; CHECK-VS-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP17]], i1 false)
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = xor <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
+; CHECK-VS-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP16]], i1 false)
 ; CHECK-VS-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
-; CHECK-VS-NEXT:    [[TMP18:%.*]] = extractelement <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = extractelement <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
 ; CHECK-VS-NEXT:    br label %[[FOR_END]]
 ; CHECK-VS:       [[FOR_END]]:
-; CHECK-VS-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[TMP18]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[TMP17]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP10]], %[[MIDDLE_BLOCK]] ]
 ; CHECK-VS-NEXT:    ret i32 [[LOAD_LCSSA]]
 ;
 entry:
@@ -258,6 +255,148 @@ for.end:
   ret i32 %load
 }
 
+define i32 @live_out_recurrence(ptr %A, i64 %n) {
+; CHECK-LABEL: define i32 @live_out_recurrence(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[VECTOR_RECUR_EXTRACT_FOR_PHI:%.*]] = extractelement <16 x i32> [[WIDE_LOAD]], i64 14
+; CHECK-NEXT:    [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <16 x i32> [[WIDE_LOAD]], i64 15
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[SCALAR_RECUR_INIT:%.*]] = phi i32 [ [[VECTOR_RECUR_EXTRACT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-NEXT:    [[VECTOR_RECUR_INIT:%.*]] = insertelement <8 x i32> poison, i32 [[SCALAR_RECUR_INIT]], i32 7
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VECTOR_RECUR:%.*]] = phi <8 x i32> [ [[VECTOR_RECUR_INIT]], %[[VEC_EPILOG_PH]] ], [ [[WIDE_MASKED_LOAD:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX2]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
+; CHECK-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP6:%.*]] = xor i1 [[TMP5]], true
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i32> [[VECTOR_RECUR]], <8 x i32> [[WIDE_MASKED_LOAD]], <8 x i32> <i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14>
+; CHECK-NEXT:    [[TMP8:%.*]] = xor <8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
+; CHECK-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v8i1(<8 x i1> [[TMP8]], i1 false)
+; CHECK-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
+; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <8 x i32> [[TMP7]], i64 [[LAST_ACTIVE_LANE]]
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[FOR_LCSSA:%.*]] = phi i32 [ [[TMP9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[FOR_LCSSA]]
+;
+; CHECK-VS-LABEL: define i32 @live_out_recurrence(
+; CHECK-VS-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 [[TMP3]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP5]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = mul nuw i32 [[TMP7]], 16
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = sub i32 [[TMP8]], 2
+; CHECK-VS-NEXT:    [[VECTOR_RECUR_EXTRACT_FOR_PHI:%.*]] = extractelement <vscale x 16 x i32> [[WIDE_LOAD]], i32 [[TMP9]]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = mul nuw i32 [[TMP10]], 16
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = sub i32 [[TMP11]], 1
+; CHECK-VS-NEXT:    [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <vscale x 16 x i32> [[WIDE_LOAD]], i32 [[TMP12]]
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[SCALAR_RECUR_INIT:%.*]] = phi i32 [ [[VECTOR_RECUR_EXTRACT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = shl nuw i64 [[TMP13]], 3
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = mul nuw i32 [[TMP15]], 8
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = sub i32 [[TMP16]], 1
+; CHECK-VS-NEXT:    [[VECTOR_RECUR_INIT:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[SCALAR_RECUR_INIT]], i32 [[TMP17]]
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[VECTOR_RECUR:%.*]] = phi <vscale x 8 x i32> [ [[VECTOR_RECUR_INIT]], %[[VEC_EPILOG_PH]] ], [ [[WIDE_MASKED_LOAD:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX2]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 4 [[TMP18]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], [[TMP14]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP20:%.*]] = xor i1 [[TMP19]], true
+; CHECK-VS-NEXT:    br i1 [[TMP20]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[TMP21:%.*]] = call <vscale x 8 x i32> @llvm.vector.splice.right.nxv8i32(<vscale x 8 x i32> [[VECTOR_RECUR]], <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], i32 1)
+; CHECK-VS-NEXT:    [[TMP22:%.*]] = xor <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
+; CHECK-VS-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP22]], i1 false)
+; CHECK-VS-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
+; CHECK-VS-NEXT:    [[TMP23:%.*]] = extractelement <vscale x 8 x i32> [[TMP21]], i64 [[LAST_ACTIVE_LANE]]
+; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[EXIT]]:
+; CHECK-VS-NEXT:    [[FOR_LCSSA:%.*]] = phi i32 [ [[TMP23]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    ret i32 [[FOR_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %for = phi i32 [ 0, %entry ], [ %l, %loop ]
+  %gep = getelementptr inbounds i32, ptr %A, i64 %iv
+  %l = load i32, ptr %gep, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, %n
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  ret i32 %for
+}
+
 define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-LABEL: define void @reversed-loop(
 ; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0]] {
@@ -362,29 +501,28 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
 ; CHECK-VS-NEXT:    [[TMP10:%.*]] = shl nuw i32 [[TMP3]], 4
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = shl nuw i32 [[TMP10]], 1
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i32 [[TMP2]], [[TMP11]]
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i32 [[TMP2]], [[TMP9]]
 ; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP2]], [[N_MOD_VF]]
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[VAL]], i64 0
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = sub i32 [[ST]], [[N_VEC]]
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = sub i32 [[ST]], [[N_VEC]]
 ; CHECK-VS-NEXT:    [[REVERSE:%.*]] = call <vscale x 16 x i32> @llvm.vector.reverse.nxv16i32(<vscale x 16 x i32> [[BROADCAST_SPLAT]])
 ; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VECTOR_BODY]]:
 ; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = sub i32 [[ST]], [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP13]]
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = zext i32 [[TMP10]] to i64
-; CHECK-VS-NEXT:    [[TMP16:%.*]] = sub nuw nsw i64 [[TMP15]], 1
-; CHECK-VS-NEXT:    [[TMP17:%.*]] = sub i64 0, [[TMP16]]
-; CHECK-VS-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[TMP14]], i64 [[TMP17]]
-; CHECK-VS-NEXT:    [[TMP19:%.*]] = sub i64 [[TMP17]], [[TMP15]]
-; CHECK-VS-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[TMP14]], i64 [[TMP19]]
-; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP18]], align 4
-; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP20]], align 4
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], [[TMP11]]
-; CHECK-VS-NEXT:    [[TMP21:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP21]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = sub i32 [[ST]], [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP12]]
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = zext i32 [[TMP10]] to i64
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = sub nuw nsw i64 [[TMP14]], 1
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = sub i64 0, [[TMP15]]
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = getelementptr inbounds i32, ptr [[TMP13]], i64 [[TMP16]]
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = sub i64 [[TMP16]], [[TMP14]]
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = getelementptr inbounds i32, ptr [[TMP13]], i64 [[TMP18]]
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP17]], align 4
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP19]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], [[TMP9]]
+; CHECK-VS-NEXT:    [[TMP20:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP20]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-VS:       [[MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP2]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
@@ -392,8 +530,8 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vscale.i32()
-; CHECK-VS-NEXT:    [[TMP23:%.*]] = shl nuw i32 [[TMP22]], 3
+; CHECK-VS-NEXT:    [[TMP21:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP22:%.*]] = shl nuw i32 [[TMP21]], 3
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[VAL]], i64 0
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT3]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
 ; CHECK-VS-NEXT:    [[REVERSE5:%.*]] = call <vscale x 8 x i32> @llvm.vector.reverse.nxv8i32(<vscale x 8 x i32> [[BROADCAST_SPLAT4]])
@@ -402,19 +540,19 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
 ; CHECK-VS-NEXT:    [[INDEX6:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP24:%.*]] = sub i32 [[ST]], [[INDEX6]]
-; CHECK-VS-NEXT:    [[TMP25:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP24]]
-; CHECK-VS-NEXT:    [[TMP26:%.*]] = zext i32 [[TMP23]] to i64
-; CHECK-VS-NEXT:    [[TMP27:%.*]] = sub nuw nsw i64 [[TMP26]], 1
-; CHECK-VS-NEXT:    [[TMP28:%.*]] = sub i64 0, [[TMP27]]
-; CHECK-VS-NEXT:    [[TMP29:%.*]] = getelementptr i32, ptr [[TMP25]], i64 [[TMP28]]
+; CHECK-VS-NEXT:    [[TMP23:%.*]] = sub i32 [[ST]], [[INDEX6]]
+; CHECK-VS-NEXT:    [[TMP24:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP23]]
+; CHECK-VS-NEXT:    [[TMP25:%.*]] = zext i32 [[TMP22]] to i64
+; CHECK-VS-NEXT:    [[TMP26:%.*]] = sub nuw nsw i64 [[TMP25]], 1
+; CHECK-VS-NEXT:    [[TMP27:%.*]] = sub i64 0, [[TMP26]]
+; CHECK-VS-NEXT:    [[TMP28:%.*]] = getelementptr i32, ptr [[TMP24]], i64 [[TMP27]]
 ; CHECK-VS-NEXT:    [[REVERSE7:%.*]] = call <vscale x 8 x i1> @llvm.vector.reverse.nxv8i1(<vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[REVERSE5]], ptr align 4 [[TMP29]], <vscale x 8 x i1> [[REVERSE7]])
-; CHECK-VS-NEXT:    [[INDEX_NEXT8]] = add i32 [[INDEX6]], [[TMP23]]
+; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[REVERSE5]], ptr align 4 [[TMP28]], <vscale x 8 x i1> [[REVERSE7]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT8]] = add i32 [[INDEX6]], [[TMP22]]
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i32(i32 [[INDEX_NEXT8]], i32 [[TMP2]])
-; CHECK-VS-NEXT:    [[TMP30:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP31:%.*]] = xor i1 [[TMP30]], true
-; CHECK-VS-NEXT:    br i1 [[TMP31]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP29:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP30:%.*]] = xor i1 [[TMP29]], true
+; CHECK-VS-NEXT:    br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    br label %[[EXIT]]
 ; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll
index 0a1eddf6e2804..de28bb59b6bab 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll
@@ -1,6 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^scalar.ph" --version 6
 ; RUN: opt -passes=loop-vectorize -enable-epilogue-vectorization=false -S < %s | FileCheck %s --check-prefixes=CHECK
 ; RUN: opt -passes=loop-vectorize -enable-epilogue-vectorization=false -tail-folding-policy=must-fold-tail -S < %s | FileCheck %s --check-prefixes=CHECK-TAILFOLD
+; RUN: opt -passes=loop-vectorize -force-vector-width=16 -epilogue-vectorization-force-VF=8 -epilogue-tail-folding-policy=prefer-fold-tail -S < %s | FileCheck %s --check-prefixes=CHECK-TAILFOLD-EPILOGUE
 
 target triple = "aarch64-none-unknown-elf"
 
@@ -80,6 +81,79 @@ define i32 @pred_reduction(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP10]]
 ;
+; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @pred_reduction(
+; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[SRC:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP7]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP8]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP9]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE5]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP10]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE5]], [[PARTIAL_REDUCE]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP12]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX6:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI7:%.*]] = phi <2 x i32> [ [[TMP13]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX6]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD8:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP14]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD8]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP15]], <8 x i1> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX6]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD9:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP17]], <8 x i1> [[TMP16]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD9]] to <8 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = select <8 x i1> [[TMP16]], <8 x i32> [[TMP18]], <8 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE10]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI7]], <8 x i32> [[TMP19]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT11]] = add i64 [[INDEX6]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT11]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = xor i1 [[TMP20]], true
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE10]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
+;
 entry:
   br label %for.body
 
@@ -184,6 +258,79 @@ define i32 @pred_reduction_sext(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP10]]
 ;
+; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @pred_reduction_sext(
+; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[SRC:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = sext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP7]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP8]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = sext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP9]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE5]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP10]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE5]], [[PARTIAL_REDUCE]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP12]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX6:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI7:%.*]] = phi <2 x i32> [ [[TMP13]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX6]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD8:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP14]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD8]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP15]], <8 x i1> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX6]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD9:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP17]], <8 x i1> [[TMP16]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = sext <8 x i8> [[WIDE_MASKED_LOAD9]] to <8 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = select <8 x i1> [[TMP16]], <8 x i32> [[TMP18]], <8 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE10]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI7]], <8 x i32> [[TMP19]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT11]] = add i64 [[INDEX6]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT11]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = xor i1 [[TMP20]], true
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE10]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
+;
 entry:
   br label %for.body
 
@@ -300,6 +447,91 @@ define i32 @pred_reduction_dotprod(ptr %a, ptr %b, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP13]]
 ;
+; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @pred_reduction_dotprod(
+; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE7:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = getelementptr i8, ptr [[TMP7]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD5:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP7]], <16 x i1> [[TMP3]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD6:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP8]], <16 x i1> [[TMP4]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD5]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = mul nuw nsw <16 x i32> [[TMP9]], [[TMP10]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP11]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP12]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD6]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = mul nuw nsw <16 x i32> [[TMP13]], [[TMP14]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP15]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE7]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP16]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE7]], [[PARTIAL_REDUCE]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP18]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX8:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT14:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI9:%.*]] = phi <2 x i32> [ [[TMP19]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX8]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD10:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP20]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD10]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP21]], <8 x i1> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP23:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX8]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD11:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP23]], <8 x i1> [[TMP22]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP24:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX8]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD12:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP24]], <8 x i1> [[TMP22]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP25:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD11]] to <8 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP26:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD12]] to <8 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP27:%.*]] = mul nuw nsw <8 x i32> [[TMP25]], [[TMP26]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP28:%.*]] = select <8 x i1> [[TMP22]], <8 x i32> [[TMP27]], <8 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE13]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI9]], <8 x i32> [[TMP28]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT14]] = add i64 [[INDEX8]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT14]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP30:%.*]] = xor i1 [[TMP29]], true
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP31:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE13]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[TMP31]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP18]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
+;
 entry:
   br label %for.body
 
@@ -422,6 +654,92 @@ define i32 @pred_sub_reduction(ptr %a, ptr %b, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP14]]
 ;
+; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @pred_sub_reduction(
+; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE7:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = getelementptr i8, ptr [[TMP7]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD5:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP7]], <16 x i1> [[TMP3]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD6:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP8]], <16 x i1> [[TMP4]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD5]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = mul nuw nsw <16 x i32> [[TMP9]], [[TMP10]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP11]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP12]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD6]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = mul nuw nsw <16 x i32> [[TMP13]], [[TMP14]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP15]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE7]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP16]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE7]], [[PARTIAL_REDUCE]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = sub i32 0, [[TMP18]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP19]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX8:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT14:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI9:%.*]] = phi <2 x i32> [ zeroinitializer, %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX8]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD10:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP20]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD10]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP21]], <8 x i1> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP23:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX8]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD11:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP23]], <8 x i1> [[TMP22]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP24:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX8]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD12:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP24]], <8 x i1> [[TMP22]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP25:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD11]] to <8 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP26:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD12]] to <8 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP27:%.*]] = mul nuw nsw <8 x i32> [[TMP25]], [[TMP26]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP28:%.*]] = select <8 x i1> [[TMP22]], <8 x i32> [[TMP27]], <8 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE13]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI9]], <8 x i32> [[TMP28]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT14]] = add i64 [[INDEX8]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT14]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP30:%.*]] = xor i1 [[TMP29]], true
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP31:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE13]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP32:%.*]] = sub i32 [[BC_MERGE_RDX]], [[TMP31]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[TMP32]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP19]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
+;
 entry:
   br label %for.body
 
@@ -543,6 +861,92 @@ define i32 @chained_pred_reduction(ptr %src, ptr noalias %src_b, ptr %cond, i64
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP13]]
 ;
+; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @chained_pred_reduction(
+; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[SRC:%.*]], ptr noalias [[SRC_B:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE8:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE9:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP7]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE:%.*]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP8]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP9]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE5:%.*]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP10]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP11]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD6:%.*]] = load <16 x i8>, ptr [[TMP11]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD7:%.*]] = load <16 x i8>, ptr [[TMP12]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = zext <16 x i8> [[WIDE_LOAD6]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE8]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[PARTIAL_REDUCE]], <16 x i32> [[TMP13]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = zext <16 x i8> [[WIDE_LOAD7]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE9]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[PARTIAL_REDUCE5]], <16 x i32> [[TMP14]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE9]], [[PARTIAL_REDUCE8]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP16]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT17:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI11:%.*]] = phi <2 x i32> [ [[TMP17]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE16:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX10]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD12:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP18]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD12]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP19]], <8 x i1> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX10]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD13:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP21]], <8 x i1> [[TMP20]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD13]] to <8 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP23:%.*]] = select <8 x i1> [[TMP20]], <8 x i32> [[TMP22]], <8 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE14:%.*]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI11]], <8 x i32> [[TMP23]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP24:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[INDEX10]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD15:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP24]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP25:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD15]] to <8 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP26:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP25]], <8 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE16]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[PARTIAL_REDUCE14]], <8 x i32> [[TMP26]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT17]] = add i64 [[INDEX10]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT17]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP27:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP28:%.*]] = xor i1 [[TMP27]], true
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP28]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE16]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2_LCSSA:%.*]] = phi i32 [ [[TMP29]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP16]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_2_LCSSA]]
+;
 entry:
   br label %for.body
 
@@ -664,6 +1068,92 @@ define i32 @reduction_before_pred(ptr %src, ptr noalias %src_b, ptr %cond, i64 %
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP12]]
 ;
+; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @reduction_before_pred(
+; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[SRC:%.*]], ptr noalias [[SRC_B:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE8:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE9:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = zext <16 x i8> [[WIDE_LOAD]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE:%.*]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP3]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = zext <16 x i8> [[WIDE_LOAD3]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE4:%.*]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP4]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP5]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD5:%.*]] = load <16 x i8>, ptr [[TMP5]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD6:%.*]] = load <16 x i8>, ptr [[TMP6]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD5]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD6]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = getelementptr i8, ptr [[TMP9]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP9]], <16 x i1> [[TMP7]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD7:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP10]], <16 x i1> [[TMP8]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = select <16 x i1> [[TMP7]], <16 x i32> [[TMP11]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE8]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[PARTIAL_REDUCE]], <16 x i32> [[TMP12]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD7]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = select <16 x i1> [[TMP8]], <16 x i32> [[TMP13]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE9]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[PARTIAL_REDUCE4]], <16 x i32> [[TMP14]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE9]], [[PARTIAL_REDUCE8]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP16]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT17:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI11:%.*]] = phi <2 x i32> [ [[TMP17]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE16:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[INDEX10]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD12:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP18]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD12]] to <8 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP19]], <8 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE13:%.*]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI11]], <8 x i32> [[TMP20]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX10]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD14:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP21]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD14]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP23:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP22]], <8 x i1> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP24:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX10]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD15:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP24]], <8 x i1> [[TMP23]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP25:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD15]] to <8 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP26:%.*]] = select <8 x i1> [[TMP23]], <8 x i32> [[TMP25]], <8 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE16]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[PARTIAL_REDUCE13]], <8 x i32> [[TMP26]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT17]] = add i64 [[INDEX10]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT17]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP27:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP28:%.*]] = xor i1 [[TMP27]], true
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP28]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE16]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2_LCSSA:%.*]] = phi i32 [ [[TMP29]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP16]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_2_LCSSA]]
+;
 entry:
   br label %for.body
 
@@ -780,6 +1270,79 @@ define i32 @pred_reduction_incoming_1(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP14]]
 ;
+; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @pred_reduction_incoming_1(
+; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[SRC:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP7]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP8]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP9]], <16 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE5]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP10]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE5]], [[PARTIAL_REDUCE]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP12]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX6:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI7:%.*]] = phi <2 x i32> [ [[TMP13]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX6]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD8:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP14]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD8]], zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP15]], <8 x i1> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX6]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD9:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP17]], <8 x i1> [[TMP16]], <8 x i8> poison)
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD9]] to <8 x i32>
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = select <8 x i1> [[TMP16]], <8 x i32> [[TMP18]], <8 x i32> zeroinitializer
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE10]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI7]], <8 x i32> [[TMP19]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT11]] = add i64 [[INDEX6]], 8
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT11]], i64 [[N]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = xor i1 [[TMP20]], true
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE10]])
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
+;
 entry:
   br label %for.body
 
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll
index 1ac73e9d6bf74..b2d5b6a3322ea 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll
@@ -1,5 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
 ; RUN: opt -passes=loop-vectorize -force-vector-width=8 -enable-epilogue-vectorization -epilogue-vectorization-force-VF=4 -S %s | FileCheck %s
+; RUN: opt -passes=loop-vectorize -force-vector-width=8 -epilogue-vectorization-force-VF=4 -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-target-supports-masked-memory-ops -S %s | FileCheck %s --check-prefix=CHECK-TAILFOLDED-EPILOGUE
 
 
 define void @dead_for(ptr %a, i64 %N) {
@@ -66,6 +68,62 @@ define void @dead_for(ptr %a, i64 %N) {
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
+; CHECK-TAILFOLDED-EPILOGUE-LABEL: define void @dead_for(
+; CHECK-TAILFOLDED-EPILOGUE-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_PH]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 7
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_BODY]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <8 x i64>, ptr [[TMP1]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP2:%.*]] = add <8 x i64> [[WIDE_LOAD]], splat (i64 10)
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store <8 x i64> [[TMP2]], ptr [[TMP1]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[MIDDLE_BLOCK]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP4:%.*]] = extractelement <8 x i64> [[WIDE_LOAD]], i64 7
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_PH]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_RND_UP:%.*]] = add i64 [[N]], 3
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP5:%.*]] = and i64 [[N_RND_UP]], 3
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_VEC2:%.*]] = sub i64 [[N_RND_UP]], [[TMP5]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TRIP_COUNT_MINUS_1:%.*]] = sub i64 [[N]], 1
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[TRIP_COUNT_MINUS_1]], i64 0
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <4 x i64> poison, i64 [[VEC_EPILOG_RESUME_VAL]], i64 0
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT3]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDUCTION:%.*]] = add nuw <4 x i64> [[BROADCAST_SPLAT4]], <i64 0, i64 1, i64 2, i64 3>
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX3:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP6:%.*]] = icmp ule <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX3]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <4 x i64> @llvm.masked.load.v4i64.p0(ptr align 4 [[TMP7]], <4 x i1> [[TMP6]], <4 x i64> poison)
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[WIDE_MASKED_LOAD]], splat (i64 10)
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    call void @llvm.masked.store.v4i64.p0(<4 x i64> [[TMP8]], ptr align 4 [[TMP7]], <4 x i1> [[TMP6]])
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX_NEXT4]] = add i64 [[INDEX3]], 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_IND_NEXT]] = add nuw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT4]], [[N_VEC2]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[TMP9]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[EXIT]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    ret void
+;
 entry:
   br label %loop
 
@@ -129,6 +187,50 @@ define i64 @for_phi_used_in_loop_and_live_out(ptr %a, i64 %N) {
 ; CHECK-NEXT:    [[RES:%.*]] = add i64 [[L_LCSSA]], [[FOR_LCSSA]]
 ; CHECK-NEXT:    ret i64 [[RES]]
 ;
+; CHECK-TAILFOLDED-EPILOGUE-LABEL: define i64 @for_phi_used_in_loop_and_live_out(
+; CHECK-TAILFOLDED-EPILOGUE-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:  [[ENTRY:.*]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_PH]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 7
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_BODY]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR:%.*]] = phi <8 x i64> [ <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 99>, %[[VECTOR_PH]] ], [ [[WIDE_LOAD:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[WIDE_LOAD]] = load <8 x i64>, ptr [[TMP1]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i64> [[VECTOR_RECUR]], <8 x i64> [[WIDE_LOAD]], <8 x i32> <i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14>
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store <8 x i64> [[TMP2]], ptr [[TMP1]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[MIDDLE_BLOCK]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR_EXTRACT_FOR_PHI:%.*]] = extractelement <8 x i64> [[WIDE_LOAD]], i64 6
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <8 x i64> [[WIDE_LOAD]], i64 7
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[SCALAR_PH]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[SCALAR_RECUR_INIT:%.*]] = phi i64 [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ], [ 99, %[[ENTRY]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[LOOP:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[LOOP]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR:%.*]] = phi i64 [ [[SCALAR_RECUR_INIT]], %[[SCALAR_PH]] ], [ [[L:%.*]], %[[LOOP]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[GEP:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L]] = load i64, ptr [[GEP]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[ADD:%.*]] = add i64 [[L]], 10
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store i64 [[FOR]], ptr [[GEP]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[EXIT]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR_LCSSA:%.*]] = phi i64 [ [[FOR]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L_LCSSA:%.*]] = phi i64 [ [[L]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[RES:%.*]] = add i64 [[L_LCSSA]], [[FOR_LCSSA]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    ret i64 [[RES]]
+;
 entry:
   br label %loop
 
@@ -191,7 +293,7 @@ define i64 @for_phi_not_used_in_loop_and_live_out(ptr %a, i64 %N) {
 ; CHECK-NEXT:    store <4 x i64> [[TMP4]], ptr [[GEP]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT6]] = add nuw i64 [[IV]], 4
 ; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT6]], [[N_VEC3]]
-; CHECK-NEXT:    br i1 [[TMP5]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP9:![0-9]+]]
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[LOOP]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[VECTOR_RECUR_EXTRACT_FOR_PHI7:%.*]] = extractelement <4 x i64> [[WIDE_LOAD5]], i64 2
 ; CHECK-NEXT:    [[VECTOR_RECUR_EXTRACT8:%.*]] = extractelement <4 x i64> [[WIDE_LOAD5]], i64 3
@@ -217,6 +319,75 @@ define i64 @for_phi_not_used_in_loop_and_live_out(ptr %a, i64 %N) {
 ; CHECK-NEXT:    [[RES:%.*]] = add i64 [[L_LCSSA]], [[FOR_LCSSA]]
 ; CHECK-NEXT:    ret i64 [[RES]]
 ;
+; CHECK-TAILFOLDED-EPILOGUE-LABEL: define i64 @for_phi_not_used_in_loop_and_live_out(
+; CHECK-TAILFOLDED-EPILOGUE-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_PH]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 7
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_BODY]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <8 x i64>, ptr [[TMP1]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP2:%.*]] = add <8 x i64> [[WIDE_LOAD]], splat (i64 10)
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store <8 x i64> [[TMP2]], ptr [[TMP1]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[MIDDLE_BLOCK]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR_EXTRACT_FOR_PHI:%.*]] = extractelement <8 x i64> [[WIDE_LOAD]], i64 6
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <8 x i64> [[WIDE_LOAD]], i64 7
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_PH]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[SCALAR_RECUR_INIT:%.*]] = phi i64 [ [[VECTOR_RECUR_EXTRACT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 99, %[[ITER_CHECK]] ], [ 99, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_RND_UP:%.*]] = add i64 [[N]], 3
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP4:%.*]] = and i64 [[N_RND_UP]], 3
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_VEC2:%.*]] = sub i64 [[N_RND_UP]], [[TMP4]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TRIP_COUNT_MINUS_1:%.*]] = sub i64 [[N]], 1
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[TRIP_COUNT_MINUS_1]], i64 0
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <4 x i64> poison, i64 [[VEC_EPILOG_RESUME_VAL]], i64 0
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT3]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDUCTION:%.*]] = add nuw <4 x i64> [[BROADCAST_SPLAT4]], <i64 0, i64 1, i64 2, i64 3>
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR_INIT:%.*]] = insertelement <4 x i64> poison, i64 [[SCALAR_RECUR_INIT]], i32 3
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX3:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR:%.*]] = phi <4 x i64> [ [[VECTOR_RECUR_INIT]], %[[VEC_EPILOG_PH]] ], [ [[WIDE_MASKED_LOAD:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP5:%.*]] = icmp ule <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX3]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD]] = call <4 x i64> @llvm.masked.load.v4i64.p0(ptr align 4 [[TMP6]], <4 x i1> [[TMP5]], <4 x i64> poison)
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP7:%.*]] = add <4 x i64> [[WIDE_MASKED_LOAD]], splat (i64 10)
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    call void @llvm.masked.store.v4i64.p0(<4 x i64> [[TMP7]], ptr align 4 [[TMP6]], <4 x i1> [[TMP5]])
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX_NEXT4]] = add i64 [[INDEX3]], 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_IND_NEXT]] = add nuw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT4]], [[N_VEC2]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[TMP8]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP9:%.*]] = shufflevector <4 x i64> [[VECTOR_RECUR]], <4 x i64> [[WIDE_MASKED_LOAD]], <4 x i32> <i32 3, i32 4, i32 5, i32 6>
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP10:%.*]] = xor <4 x i1> [[TMP5]], splat (i1 true)
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP10]], i1 false)
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 [[LAST_ACTIVE_LANE]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[EXIT]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR_LCSSA:%.*]] = phi i64 [ [[TMP11]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L_LCSSA:%.*]] = phi i64 [ [[TMP12]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[RES:%.*]] = add i64 [[L_LCSSA]], [[FOR_LCSSA]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    ret i64 [[RES]]
+;
 entry:
   br label %loop
 
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index cff4c925366ab..d77ce077c06f0 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -14,6 +14,8 @@
 
 ; RUN: %{cmd} -force-vector-width=8 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALID-VFs
 
+; RUN: %{cmd} -force-vector-width=8 -epilogue-vectorization-force-VF=16 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALID-BIGER-EPILOGUE
+
 ; RUN: %{cmd} -force-vector-width=16 -epilogue-vectorization-force-VF=8 -enable-early-exit-vectorization-with-side-effects \
 ; RUN: < %s 2>&1 | FileCheck %s --check-prefix=CHECK-DISABLED-EARLY-EXIT
 
@@ -26,11 +28,6 @@
 ; RUN: %{cmd} -force-vector-width=16 -epilogue-vectorization-force-VF=8 \
 ; RUN: -enable-interleaved-mem-accesses=true < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALID-INTERLEAVE
 
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize --disable-output -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -vectorize-scev-check-threshold=0 < %s 2>&1 | FileCheck %s \
-; RUN: --check-prefix=CHECK-NO-VPLANS
-
-
 define void @test_epilogue_tf(ptr %A, i64 %n, i8 %val) {
 ; CHECK-LABEL: LV: Checking a loop in 'test_epilogue_tf'
 ; CHECK: LV: epilogue tail-folding is enabled
@@ -50,6 +47,9 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i8 %val) {
 ; CHECK-ALIAS-MASK-LABEL: Checking a loop in 'test_epilogue_tf'
 ; CHECK-ALIAS-MASK: remark: <unknown>:0:0: Epilogue tail-folding is not supported with alias masking
 ;
+; CHECK-INVALID-BIGER-EPILOGUE-LABEL: Checking a loop in 'test_epilogue_tf'
+; CHECK-INVALID-BIGER-EPILOGUE: remark: <unknown>:0:0: For now, epilogue tail-folding can't be applied when VF of the main loop <= VF of the epilogue
+
 entry:
   br label %for.body
 
@@ -71,7 +71,6 @@ define void @test_no_iterations_left(ptr %A) {
 ; CHECK-LABEL: LV: Checking a loop in 'test_no_iterations_left'
 ; CHECK: LV: epilogue tail-folding is enabled
 ; CHECK: LV: This case of epilogue loop can't be tail-folded.
-; CHECK: LV: Applying epilogue tail-folding failed, disable it.
 ;
 entry:
   br label %for.body
@@ -279,33 +278,6 @@ for.end:
   ret void
 }
 
-; Can't build a valid vplan for this case because too many SCEV checks needed,
-; more than the specfied limit.
-define i64 @test_no_vplan_built(ptr %dst, i64 %n) {
-; CHECK-NO-VPLANS-LABEL: Checking a loop in 'test_no_vplan_built'
-; CHECK-NO-VPLANS: LV: epilogue tail-folding is enabled
-; CHECK-NO-VPLANS: LV: no vplans have been built for main loop VF, bail out of epilogue tail-folding
-;
-entry:
-  br label %loop
-
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %dead.iv = phi i16 [ 0, %entry ], [ %dead.iv.next, %loop ]
-  %prev = phi i64 [ 0, %entry ], [ %ext, %loop ]
-  %iv.next = add nuw nsw i64 %iv, 1
-  %dead.iv.next = add i16 %dead.iv, 1
-  %ext = zext i16 %dead.iv.next to i64
-  %gep = getelementptr inbounds i64, ptr %dst, i64 %prev
-  store i64 %iv, ptr %gep, align 8
-  %cmp = icmp slt i64 %iv.next, %n
-  br i1 %cmp, label %loop, label %exit
-
-exit:
-  %result = phi i64 [ %ext, %loop ]
-  ret i64 %result
-}
-
 define void @test_outer_loop(ptr %A, i64 %m) {
 ; CHECK-OUTER-LOOP-LABEL: Checking a loop in 'test_outer_loop'
 ; CHECK-OUTER-LOOP: remark: <unknown>:0:0: Epilogue tail-folding is not supported for outer loop

>From 47cda6a2a08a7a618f683f70e2e40fb928cc0132 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Sat, 5 Sep 2026 04:25:11 +0100
Subject: [PATCH 14/25] format

---
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 11 ++++++-----
 1 file changed, 6 insertions(+), 5 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 12dcca30044bd..cd91dfd22c122 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5685,9 +5685,10 @@ getRecordedExecutionFrequency(const VPBasicBlock *VPBB) {
 }
 #endif
 
-InstructionCost LoopVectorizationPlanner::cost(
-    VPlan &Plan, ElementCount VF, VPRegisterUsage *RU,
-    LoopVectorizationCostModel &EnabledCM) const {
+InstructionCost
+LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
+                               VPRegisterUsage *RU,
+                               LoopVectorizationCostModel &EnabledCM) const {
   VPCostContext CostCtx(*TLI, Plan, EnabledCM, Config,
                         /*ReusePrintingSlotTracker=*/true);
   InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
@@ -8323,8 +8324,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   VPlan &BestPlan = *BestPlanPtr;
   // Consider vectorizing the epilogue too if it's profitable.
-  std::unique_ptr<VPlan> EpiPlan = LVP.selectBestEpiloguePlan(
-      BestPlan, VF.Width, IC, ScalarEpilogueAllowed);
+  std::unique_ptr<VPlan> EpiPlan =
+      LVP.selectBestEpiloguePlan(BestPlan, VF.Width, IC, ScalarEpilogueAllowed);
   bool HasBranchWeights =
       hasBranchWeightMD(*L->getLoopLatch()->getTerminator());
   if (EpiPlan) {

>From 9b086f816844d3a4df114812caceec7737ab1f8e Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Sat, 5 Sep 2026 06:52:50 +0100
Subject: [PATCH 15/25] exclude partial-alias

---
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 1 -
 1 file changed, 1 deletion(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index cd91dfd22c122..471b42fee1087 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -8371,7 +8371,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     SmallVector<Instruction *> InstsToMove = preparePlanForEpilogueVectorLoop(
         BestMainPlan, BestEpiPlan, L, ExpandedSCEVs, EPI, LVP, Config,
         *PSE.getSE(), ResumeValues);
-    LVP.attachRuntimeChecks(BestEpiPlan, Checks, HasBranchWeights);
     RUN_VPLAN_PASS(VPlanTransforms::simplifyLiveInsWithSCEV, BestEpiPlan, PSE);
     // Save the status of epilogue tail-folding:
     const bool IsTailFolded = BestEpiPlan.hasTailFolded();

>From a0fb982071507028b22131bae8cf382f3b7a772c Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Wed, 9 Sep 2026 15:46:20 +0100
Subject: [PATCH 16/25] resolve review comments and add extra coverage tests

---
 .../Vectorize/LoopVectorizationPlanner.h      |   8 -
 .../Transforms/Vectorize/LoopVectorize.cpp    | 144 ++++-------
 .../Transforms/Vectorize/VPlanLowering.cpp    |  34 ++-
 .../AArch64/fold-epilogue-tail-reductions.ll  |  96 ++++++-
 .../AArch64/fold-epilogue-tail.ll             | 241 ++++++++++++++++--
 .../AArch64/partial-reduce-with-predicate.ll  | 214 +++++++++++++---
 ...g-vectorization-fixed-order-recurrences.ll |  32 ++-
 .../LoopVectorize/fold-epilogue-tail.ll       |  36 +--
 8 files changed, 613 insertions(+), 192 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 291c620bf4b74..d870f049162db 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -888,10 +888,6 @@ class LoopVectorizationPlanner {
   /// The profitability analysis. Cleared after making cost based decisions.
   std::unique_ptr<LoopVectorizationCostModel> CM;
 
-  /// The profitability analysis for epilogue tail-folding.
-  /// Cleared after making cost based decisions.
-  std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM;
-
   /// VF selection state independent of cost-modeling decisions.
   VFSelectionContext &Config;
 
@@ -935,7 +931,6 @@ class LoopVectorizationPlanner {
       Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
       const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
       std::unique_ptr<LoopVectorizationCostModel> CM,
-      std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM,
       VFSelectionContext &Config, InterleavedAccessInfo &IAI,
       PredicatedScalarEvolution &PSE, OptimizationRemarkEmitter *ORE,
       std::function<const BranchProbabilityInfo &()> GetBPI);
@@ -951,9 +946,6 @@ class LoopVectorizationPlanner {
   /// Destroy the cost model.
   void clearCostModel();
 
-  /// Destroy the epilogue tail-folding cost model.
-  void clearEpilogueTfCM();
-
   /// Build VPlans for the specified \p UserVF and \p UserIC if they are
   /// non-zero or all applicable candidate VFs otherwise. If vectorization and
   /// interleaving should be avoided up-front, no plans are generated.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 471b42fee1087..af50d338856a4 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3402,7 +3402,7 @@ static bool hasFindLastReductionPhi(VPlan &Plan) {
 static EpilogueLowering getEpilogueTailLowering(
     const LoopVectorizationCostModel &MainCM, const Loop *L,
     OptimizationRemarkEmitter *ORE, const LoopVectorizationLegality &LVL,
-    const LoopVectorizeHints &Hints, TargetTransformInfo *TTI) {
+    const LoopVectorizeHints &Hints, const TargetTransformInfo *TTI) {
   // Epilogue TF is only enabled when explicitly requested via command line.
   if (!EpilogueTailFoldingPolicy.getNumOccurrences() ||
       EpilogueTailFoldingPolicy != TailFoldingPolicyTy::PreferFoldTail)
@@ -5422,7 +5422,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
       if (!VPlans.empty() && (VPlans.front()->getSingleVF() == UserVF) &&
           (UserVF.isScalar() ||
            cost(*VPlans.front(), UserVF, /*RU=*/nullptr, *CM).isValid())) {
-        // Plan for epilogue only if we succeeded in building main loop vplan.
+        // Plan for epilogue only if we succeeded in building main loop Vplan.
         // Try to plan for tail-folded epilogue if it's enabled/doable,
         // otherwise plan for unpredicated epilogue:
         if (!planForEpilogueTF()) {
@@ -5465,26 +5465,44 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 }
 
 bool LoopVectorizationPlanner::planForEpilogueTF() {
-  if (!EpilogueTfCM)
+  EpilogueLowering EpilogueTailLoweringStatus =
+      getEpilogueTailLowering(*CM, OrigLoop, ORE, *Legal, Config.getHints(),
+                              &TTI);
+  if (EpilogueTailLoweringStatus !=
+      EpilogueLowering::CM_EpilogueNotNeededFoldTail)
     return false;
-  assert(EpilogueTfCM->preferTailFoldedLoop() &&
+  LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
+
+  bool UseInterleaved = TTI.enableInterleavedAccessVectorization();
+  if (EnableInterleavedMemAccesses.getNumOccurrences() > 0)
+    UseInterleaved = EnableInterleavedMemAccesses;
+
+  InterleavedAccessInfo EpilogueTfCMIAI(PSE, OrigLoop, DT, LI, Legal->getLAI(),
+                                        Config.OptForSize);
+  if (UseInterleaved)
+    EpilogueTfCMIAI.analyzeInterleaving(useMaskedInterleavedAccesses(TTI));
+  LoopVectorizationCostModel EpilogueTfCM(
+      EpilogueTailLoweringStatus, OrigLoop, PSE, LI, Legal, TTI, TLI, CM->AC,
+      ORE, CM->GetBFI, CM->TheFunction, EpilogueTfCMIAI, Config);
+
+  assert(EpilogueTfCM.preferTailFoldedLoop() &&
          "Epilogue tail-folding is expected to be enabled");
 
   LLVM_DEBUG(dbgs() << "LV: plan for tail-folded epilogue\n");
 
-  EpilogueTfCM->ValuesToIgnore.insert_range(CM->ValuesToIgnore);
-  EpilogueTfCM->VecValuesToIgnore.insert_range(CM->VecValuesToIgnore);
+  EpilogueTfCM.ValuesToIgnore.insert_range(CM->ValuesToIgnore);
+  EpilogueTfCM.VecValuesToIgnore.insert_range(CM->VecValuesToIgnore);
 
   FixedScalableVFPair MaxFactors =
-      EpilogueTfCM->computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
-  if (!MaxFactors || !EpilogueTfCM->foldTailByMasking()) {
+      EpilogueTfCM.computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
+  if (!MaxFactors || !EpilogueTfCM.foldTailByMasking()) {
     // Cases that should not to be vectorized or tail-folded.
     reportVectorizationInfo("This case of epilogue loop can't be tail-folded",
                             "InvalidTailFoldedEpilogue", ORE, OrigLoop);
     return false;
   }
 
-  auto VPlan1 = tryToBuildVPlan1(*EpilogueTfCM);
+  auto VPlan1 = tryToBuildVPlan1(EpilogueTfCM);
 
   // If we're here, the main loop's initial VPlan was built successfully.
   // Building one for the tail-folded loop should therefore also succeed, since
@@ -5502,35 +5520,40 @@ bool LoopVectorizationPlanner::planForEpilogueTF() {
     LLVM_DEBUG(
         dbgs() << "LV: Invalidate all interleaved groups due to fold-tail by "
                   "masking which requires masked-interleaved support.\n");
-    if (EpilogueTfCM->InterleaveInfo.invalidateGroups())
+    if (EpilogueTfCM.InterleaveInfo.invalidateGroups())
       // Invalidating interleave groups also requires invalidating all decisions
       // based on them, which includes widening decisions and uniform and scalar
       // values.
-      EpilogueTfCM->invalidateCostModelingDecisions();
+      EpilogueTfCM.invalidateCostModelingDecisions();
   }
   Legal->prepareToFoldTailByMasking();
 
   // Collect the instructions (and their associated costs) that will be more
   // profitable to scalarize.
-  EpilogueTfCM->collectNonVectorizedAndSetWideningDecisions(
+  EpilogueTfCM.collectNonVectorizedAndSetWideningDecisions(
       EpilogueVectorizationForceVF);
 
   size_t NumPlansBefore = VPlans.size();
   buildVPlans(*VPlan1, EpilogueVectorizationForceVF,
-              EpilogueVectorizationForceVF, *EpilogueTfCM);
+              EpilogueVectorizationForceVF, EpilogueTfCM);
 
-  // Check that a vplan is successfully built:
-  if (VPlans.size() == NumPlansBefore ||
-      VPlans.back()->getSingleVF() != EpilogueVectorizationForceVF ||
+  // Check that a Vplan is successfully built:
+  if (VPlans.size() == NumPlansBefore) {
+    reportVectorizationInfo("Failed to build tail-folded epilogue VPlan",
+                            "InvalidTailFoldedEpilogue", ORE, OrigLoop);
+    return false;
+  }
+  if (VPlans.back()->getSingleVF() != EpilogueVectorizationForceVF ||
       !VPlans.back()->hasTailFolded()) {
     reportVectorizationInfo(
         "Failed to build a valid tail-folded epilogue VPlan",
         "InvalidTailFoldedEpilogue", ORE, OrigLoop);
+    VPlans.pop_back();
     return false;
   }
 
   if (!cost(*VPlans.back(), EpilogueVectorizationForceVF, /*RU=*/nullptr,
-            *EpilogueTfCM)
+            EpilogueTfCM)
            .isValid()) {
     VPlans.pop_back();
     reportVectorizationInfo("This case of epilogue loop can't be tail-folded "
@@ -5849,21 +5872,18 @@ LoopVectorizationPlanner::computeBestVF() {
 LoopVectorizationPlanner::LoopVectorizationPlanner(
     Loop *L, LoopInfo *LI, DominatorTree *DT, const TargetLibraryInfo *TLI,
     const TargetTransformInfo &TTI, LoopVectorizationLegality *Legal,
-    std::unique_ptr<LoopVectorizationCostModel> CM,
-    std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM,
-    VFSelectionContext &Config, InterleavedAccessInfo &IAI,
-    PredicatedScalarEvolution &PSE, OptimizationRemarkEmitter *ORE,
+    std::unique_ptr<LoopVectorizationCostModel> CM, VFSelectionContext &Config,
+    InterleavedAccessInfo &IAI, PredicatedScalarEvolution &PSE,
+    OptimizationRemarkEmitter *ORE,
     std::function<const BranchProbabilityInfo &()> GetBPI)
     : OrigLoop(L), LI(LI), DT(DT), TLI(TLI), TTI(TTI), Legal(Legal),
-      CM(std::move(CM)), EpilogueTfCM(std::move(EpilogueTfCM)), Config(Config),
-      IAI(IAI), PSE(PSE), ORE(ORE), GetBPI(GetBPI) {}
+      CM(std::move(CM)), Config(Config), IAI(IAI), PSE(PSE), ORE(ORE),
+      GetBPI(GetBPI) {}
 
 LoopVectorizationPlanner::~LoopVectorizationPlanner() = default;
 
 void LoopVectorizationPlanner::clearCostModel() { CM.reset(); }
 
-void LoopVectorizationPlanner::clearEpilogueTfCM() { EpilogueTfCM.reset(); }
-
 DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
     ElementCount BestVF, unsigned BestUF, VPlan &BestVPlan,
     InnerLoopVectorizer &ILV, DominatorTree *DT,
@@ -5891,9 +5911,8 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
                    BestVPlan, BestVF, VScale);
   }
 
-  const bool IsTailFolded = BestVPlan.hasTailFolded();
   if (vputils::findIncomingAliasMask(BestVPlan)) {
-    assert(IsTailFolded && "Expected tail folding to be enabled");
+    assert(BestVPlan.hasTailFolded() && "Expected tail folding to be enabled");
     RUN_VPLAN_PASS(VPlanTransforms::materializeAliasMaskCheckBlock, BestVPlan,
                    *Legal->getRuntimePointerChecking()->getDiffChecks(),
                    HasBranchWeights);
@@ -5933,6 +5952,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
   RUN_VPLAN_PASS(VPlanTransforms::convertEVLExitCond, BestVPlan);
   // Regions are dissolved after optimizing for VF and UF, which completely
   // removes unneeded loop regions first.
+  const bool HasTailFolded = BestVPlan.hasTailFolded();
   RUN_VPLAN_PASS(VPlanTransforms::dissolveLoopRegions, BestVPlan);
   // Expand BranchOnTwoConds after dissolution, when latch has direct access to
   // its successors.
@@ -5952,7 +5972,7 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
   assert((LI->getUniqueLatchExitBlock(*OrigLoop) || RequiresScalarEpilogue) &&
          "loops not exiting via the latch without required epilogue?");
   RUN_VPLAN_PASS(VPlanTransforms::materializeVectorTripCount, BestVPlan,
-                 VectorPH, IsTailFolded, RequiresScalarEpilogue,
+                 VectorPH, HasTailFolded, RequiresScalarEpilogue,
                  &BestVPlan.getVFxUF(), MaxRuntimeStep);
   RUN_VPLAN_PASS(VPlanTransforms::materializeFactors, BestVPlan, VectorPH,
                  BestVF);
@@ -7848,39 +7868,6 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
   for (PHINode &Phi : make_early_inc_range(VecEpiloguePreHeader->phis()))
     if (Phi.use_empty())
       Phi.eraseFromParent();
-
-  if (IsEpilogueTfEnabled) {
-    // The epilogue vector loop is tail-folded, so it can safely handle
-    // any remaining iterations, including zero, via masking.
-    // vec.epilog.iter.check's own min-iters check was therefore built with a
-    // compile-time-known-false condition (see
-    // addMinimumVectorEpilogueIterationCheck) that never needs to bail out to
-    // a scalar remainder. Fold it into an unconditional branch into the
-    // vector epilogue preheader.
-    auto *Br =
-        cast<CondBrInst>(VecEpilogueIterationCountCheck->getTerminator());
-    [[maybe_unused]] auto *CondC = dyn_cast<ConstantInt>(Br->getCondition());
-    assert(CondC && CondC->isZero() &&
-           "expected vec.epilog.iter.check's branch condition to be a "
-           "compile-time false constant when the epilogue is tail-folded");
-    BasicBlock *DeadSucc = Br->getSuccessor(0);
-    UncondBrInst::Create(VecEpiloguePreHeader, Br->getIterator());
-    Br->eraseFromParent();
-    DTU.applyUpdates(
-        {{DominatorTree::Delete, VecEpilogueIterationCountCheck, DeadSucc}});
-
-    if (!SCEVCheckBlock && !MemCheckBlock) {
-      // Delete the scalar loop as it's dead right now.
-      assert(pred_empty(ScalarPH) &&
-             "scalar preheader should have no predecessors left");
-      SmallVector<BasicBlock *> Blocks(L->block_begin(), L->block_end());
-      Blocks.push_back(ScalarPH);
-      LI->erase(L);
-      for (auto *BB : Blocks)
-        LI->removeBlock(BB);
-      DeleteDeadBlocks(Blocks, &DTU);
-    }
-  }
 }
 
 bool LoopVectorizePass::processLoop(Loop *L) {
@@ -8068,33 +8055,12 @@ bool LoopVectorizePass::processLoop(Loop *L) {
   // Use the cost model.
   VFSelectionContext Config(*TTI, &LVL, L, *F, PSE, DB, ORE, &Hints,
                             OptForSize);
-  auto CM = std::make_unique<LoopVectorizationCostModel>(
-      SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config);
-
-  // Setup the epilogue tail-folding CM. Only built when tail-folding the
-  // epilogue is actually a candidate, to avoid the cost of an extra
-  // InterleavedAccessInfo scan and LoopVectorizationCostModel construction
-  // for the common case where this (experimental, off-by-default) feature
-  // isn't in use.
-  EpilogueLowering EpilogueTailLoweringStatus =
-      getEpilogueTailLowering(*CM, L, ORE, LVL, Hints, TTI);
-  std::optional<InterleavedAccessInfo> EpilogueTfCMIAI;
-  std::unique_ptr<LoopVectorizationCostModel> EpilogueTfCM;
-  if (EpilogueTailLoweringStatus ==
-      EpilogueLowering::CM_EpilogueNotNeededFoldTail) {
-    LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
-    EpilogueTfCMIAI.emplace(PSE, L, DT, LI, LVL.getLAI(), OptForSize);
-    if (UseInterleaved)
-      EpilogueTfCMIAI->analyzeInterleaving(useMaskedInterleavedAccesses(*TTI));
-    EpilogueTfCM = std::make_unique<LoopVectorizationCostModel>(
-        EpilogueTailLoweringStatus, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE,
-        GetBFI, F, *EpilogueTfCMIAI, Config);
-  }
-
   // Use the planner for vectorization.
-  LoopVectorizationPlanner LVP(L, LI, DT, TLI, *TTI, &LVL, std::move(CM),
-                               std::move(EpilogueTfCM), Config, IAI, PSE,
-                               ORE, GetBPI);
+  LoopVectorizationPlanner LVP(
+      L, LI, DT, TLI, *TTI, &LVL,
+      std::make_unique<LoopVectorizationCostModel>(
+          SEL, L, PSE, LI, &LVL, *TTI, TLI, AC, ORE, GetBFI, F, IAI, Config),
+      Config, IAI, PSE, ORE, GetBPI);
 
   // Get user vectorization factor and interleave count.
   ElementCount UserVF = Hints.getWidth();
@@ -8108,10 +8074,6 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Plan how to best vectorize.
   LVP.plan(UserVF, UserIC);
-  // Right now, after planning, the epilogue tail-folding CM is not needed
-  // anymore. Clear it.
-  LVP.clearEpilogueTfCM();
-
   auto [VF, BestPlanPtr] = LVP.computeBestVF();
   unsigned IC = 1;
 
diff --git a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
index 5a9124b8eea3a..b724084bf2a97 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
@@ -49,29 +49,30 @@ void VPlanTransforms::replaceWideCanonicalIVWithWideIV(
   VPIRValue *StartValue = nullptr;
   // VPWidenCanonicalIVRecipe is either a direct user of CanonicalIV or
   // Add (CanonicalIV, resumeValue) (like the case for tail-folded epilogue).
-  for (auto *User : LoopRegion->getCanonicalIV()->users()) {
+  auto *IV = LoopRegion->getCanonicalIV();
+  for (auto *User : IV->users()) {
     if (isa<VPWidenCanonicalIVRecipe>(User)) {
       WideCanIV = cast<VPWidenCanonicalIVRecipe>(User);
       StartValue = Plan.getZero(WideCanIV->getScalarType());
       break;
     }
-    if (isa<VPInstruction>(User)) {
-      auto *UserInstr = cast<VPInstruction>(User);
-      auto *It = find_if(UserInstr->users(), IsaPred<VPWidenCanonicalIVRecipe>);
-      if (It != UserInstr->user_end()) {
+    if (auto *UI = dyn_cast<VPInstruction>(User)) {
+      auto *It = find_if(UI->users(), IsaPred<VPWidenCanonicalIVRecipe>);
+      if (It != UI->user_end()) {
         WideCanIV = cast<VPWidenCanonicalIVRecipe>(*It);
-        match(UserInstr, m_Add(m_VPValue(), m_VPIRValue(StartValue)));
-        assert(StartValue &&
-               "WIDEN-CANONICAL-INDUCTION is only expected to be reached "
-               "through the canonical IV directly, or through a single 'add "
-               "CanonicalIV, StartValue' introduced for epilogue-loop resume "
-               "values; found a different pattern here");
-        if (!StartValue)
+        if (!match(UI, m_c_Add(m_Specific(IV), m_VPIRValue(StartValue))) ||
+            !StartValue) {
+          assert(StartValue &&
+                 "WIDEN-CANONICAL-INDUCTION is only expected to be reached "
+                 "through the canonical IV directly, or through a single 'add "
+                 "CanonicalIV, StartValue' introduced for epilogue-loop resume "
+                 "values; found a different pattern here");
           return;
+        }
       }
     }
   }
-  if (!WideCanIV)
+  if (!WideCanIV || !StartValue)
     return;
 
   Type *CanIVTy = WideCanIV->getScalarType();
@@ -579,7 +580,12 @@ void VPlanTransforms::convertToConcreteRecipes(VPlan &Plan) {
       }
 
       if (auto *WideCanIV = dyn_cast<VPWidenCanonicalIVRecipe>(&R)) {
-        VPValue *CanIV = WideCanIV->getCanonicalIV();
+        VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
+        if (!LoopRegion)
+          continue;
+        VPValue *CanIV = LoopRegion->getCanonicalIV();
+        if (!CanIV)
+          continue;
         Type *CanIVTy = CanIV->getScalarType();
         VPValue *Step = WideCanIV->getStepValue();
         if (!Step) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll
index 1c9c05884b7d4..4323230e2b525 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll
@@ -41,7 +41,7 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -64,8 +64,19 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP11]])
 ; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[ADD:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT:    [[LOAD:%.*]] = load i32, ptr [[GEP]], align 1
+; CHECK-NEXT:    [[ADD]] = add i32 [[LOAD]], [[RED]]
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
+; CHECK-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP14]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[ADD]], %[[LOOP]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[TMP14]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[ADD_LCSSA]]
 ;
 ; CHECK-VS-LABEL: define i32 @add_redc(
@@ -104,7 +115,7 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -129,8 +140,19 @@ define i32 @add_redc(ptr %src, i64 %n) {
 ; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    [[TMP20:%.*]] = call i32 @llvm.vector.reduce.add.nxv8i32(<vscale x 8 x i32> [[TMP17]])
 ; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
+; CHECK-VS:       [[LOOP]]:
+; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-VS-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[ADD:%.*]], %[[LOOP]] ]
+; CHECK-VS-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[IV]]
+; CHECK-VS-NEXT:    [[LOAD:%.*]] = load i32, ptr [[GEP]], align 1
+; CHECK-VS-NEXT:    [[ADD]] = add i32 [[LOAD]], [[RED]]
+; CHECK-VS-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-VS-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
+; CHECK-VS-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK-VS:       [[EXIT]]:
-; CHECK-VS-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[TMP20]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[ADD]], %[[LOOP]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ], [ [[TMP20]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-VS-NEXT:    ret i32 [[ADD_LCSSA]]
 ;
 entry:
@@ -183,7 +205,7 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -207,8 +229,19 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[TMP13:%.*]] = call i32 @llvm.vector.reduce.umax.v8i32(<8 x i32> [[TMP10]])
 ; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MAX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT:    [[LOAD:%.*]] = load i32, ptr [[GEP]], align 1
+; CHECK-NEXT:    [[MAX]] = call i32 @llvm.umax.i32(i32 [[LOAD]], i32 [[RED]])
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
+; CHECK-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP13]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[MAX]], %[[LOOP]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[TMP13]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[MAX_LCSSA]]
 ;
 ; CHECK-VS-LABEL: define i32 @max_redc(
@@ -247,7 +280,7 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -273,8 +306,19 @@ define i32 @max_redc(ptr %src, i64 %n) {
 ; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    [[TMP19:%.*]] = call i32 @llvm.vector.reduce.umax.nxv8i32(<vscale x 8 x i32> [[TMP16]])
 ; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
+; CHECK-VS:       [[LOOP]]:
+; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-VS-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MAX:%.*]], %[[LOOP]] ]
+; CHECK-VS-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[IV]]
+; CHECK-VS-NEXT:    [[LOAD:%.*]] = load i32, ptr [[GEP]], align 1
+; CHECK-VS-NEXT:    [[MAX]] = call i32 @llvm.umax.i32(i32 [[LOAD]], i32 [[RED]])
+; CHECK-VS-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-VS-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
+; CHECK-VS-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK-VS:       [[EXIT]]:
-; CHECK-VS-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[TMP19]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[MAX]], %[[LOOP]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ], [ [[TMP19]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-VS-NEXT:    ret i32 [[MAX_LCSSA]]
 ;
 entry:
@@ -455,7 +499,7 @@ define i32 @any-of(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -483,8 +527,20 @@ define i32 @any-of(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[TMP19:%.*]] = freeze i1 [[TMP18]]
 ; CHECK-NEXT:    [[RDX_SELECT7:%.*]] = select i1 [[TMP19]], i32 1, i32 0
 ; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SELECT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT:    [[LOAD:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-NEXT:    [[ICMP:%.*]] = icmp eq i8 [[LOAD]], 0
+; CHECK-NEXT:    [[SELECT]] = select i1 [[ICMP]], i32 1, i32 [[RED]]
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
+; CHECK-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[SELECT_LCSSA:%.*]] = phi i32 [ [[RDX_SELECT7]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    [[SELECT_LCSSA:%.*]] = phi i32 [ [[SELECT]], %[[LOOP]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ [[RDX_SELECT7]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[SELECT_LCSSA]]
 ;
 ; CHECK-VS-LABEL: define i32 @any-of(
@@ -527,7 +583,7 @@ define i32 @any-of(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -557,8 +613,20 @@ define i32 @any-of(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[TMP25:%.*]] = freeze i1 [[TMP24]]
 ; CHECK-VS-NEXT:    [[RDX_SELECT7:%.*]] = select i1 [[TMP25]], i32 1, i32 0
 ; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
+; CHECK-VS:       [[LOOP]]:
+; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-VS-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SELECT:%.*]], %[[LOOP]] ]
+; CHECK-VS-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-VS-NEXT:    [[LOAD:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-VS-NEXT:    [[ICMP:%.*]] = icmp eq i8 [[LOAD]], 0
+; CHECK-VS-NEXT:    [[SELECT]] = select i1 [[ICMP]], i32 1, i32 [[RED]]
+; CHECK-VS-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-VS-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
+; CHECK-VS-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK-VS:       [[EXIT]]:
-; CHECK-VS-NEXT:    [[SELECT_LCSSA:%.*]] = phi i32 [ [[RDX_SELECT7]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    [[SELECT_LCSSA:%.*]] = phi i32 [ [[SELECT]], %[[LOOP]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ [[RDX_SELECT7]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-VS-NEXT:    ret i32 [[SELECT_LCSSA]]
 ;
 entry:
@@ -621,7 +689,7 @@ define i64 @arg_min_first_index(ptr %arr, i64 %n, i64 %start) {
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
 ; CHECK-NEXT:    [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
-; CHECK-NEXT:    br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF11:![0-9]+]]
+; CHECK-NEXT:    br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[TMP6]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[START]], %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -729,7 +797,7 @@ define i64 @arg_min_first_index(ptr %arr, i64 %n, i64 %start) {
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
 ; CHECK-VS-NEXT:    [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], [[TMP1]]
-; CHECK-VS-NEXT:    br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF11:![0-9]+]]
+; CHECK-VS-NEXT:    br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[TMP9]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[START]], %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 3cee861a822b4..a610939954df2 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -1,12 +1,31 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; REQUIRES: asserts
 
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width=16 -epilogue-vectorization-force-VF=8 -mattr=+sve %s | FileCheck %s
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -mattr=+sve \
+; RUN: -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width=16 \
+; RUN: -epilogue-vectorization-force-VF=8 %s | FileCheck %s
 
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail -force-vector-width="vscale x 16" -epilogue-vectorization-force-VF="vscale x 8" -mattr=+sve %s | FileCheck %s --check-prefix=CHECK-VS
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -mattr=+sve \
+; RUN: -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width="vscale x 16" \
+; RUN: -epilogue-vectorization-force-VF="vscale x 8" %s | FileCheck %s \
+; RUN: --check-prefix=CHECK-VS
 
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail -debug-only=loop-vectorize,vectorutils --disable-output \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALIDATE-INTERLEAVE
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize \
+; RUN: -epilogue-tail-folding-policy=prefer-fold-tail --disable-output \
+; RUN: -force-vector-width=4 -epilogue-vectorization-force-VF="vscale x 2" \
+; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s \
+; RUN: --check-prefix=CHECK-INVALID-COSTS
+
+; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
+; RUN: -force-vector-width=8 -epilogue-vectorization-force-VF=4 \
+; RUN: -force-target-supports-masked-memory-ops -S %s | FileCheck %s \
+; RUN: --check-prefix=CHECK-INDUCTION
+
+; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize,vectorutils \
+; RUN: -epilogue-tail-folding-policy=prefer-fold-tail --disable-output \
+; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 \
+; RUN: | FileCheck %s --check-prefix=CHECK-INVALIDATE-INTERLEAVE
 
 target triple = "aarch64-linux-gnu"
 
@@ -38,7 +57,7 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
@@ -57,6 +76,15 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i32 [[VAL]], ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[EXITCOND]], label %[[FOR_BODY]], label %[[EXIT]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
@@ -91,7 +119,7 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-VS-NEXT:    [[TMP7:%.*]] = call i64 @llvm.vscale.i64()
@@ -112,6 +140,15 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-VS-NEXT:    br i1 [[TMP11]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-VS-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-VS:       [[FOR_BODY]]:
+; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-VS-NEXT:    store i32 [[VAL]], ptr [[ARRAYIDX]], align 4
+; CHECK-VS-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-VS-NEXT:    [[EXITCOND:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
+; CHECK-VS-NEXT:    br i1 [[EXITCOND]], label %[[FOR_BODY]], label %[[EXIT]]
 ; CHECK-VS:       [[EXIT]]:
 ; CHECK-VS-NEXT:    ret void
 ;
@@ -156,7 +193,7 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
@@ -177,8 +214,17 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
 ; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <8 x i32> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
 ; CHECK-NEXT:    br label %[[FOR_END]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT:    [[LOAD:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[FOR_END]], label %[[LOOP]]
 ; CHECK:       [[FOR_END]]:
-; CHECK-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[TMP9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP4]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[LOAD]], %[[LOOP]] ], [ [[TMP4]], %[[MIDDLE_BLOCK]] ], [ [[TMP9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[LOAD_LCSSA]]
 ;
 ; CHECK-VS-LABEL: define i32 @live-out(
@@ -213,7 +259,7 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i64 @llvm.vscale.i64()
@@ -236,8 +282,17 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
 ; CHECK-VS-NEXT:    [[TMP17:%.*]] = extractelement <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
 ; CHECK-VS-NEXT:    br label %[[FOR_END]]
+; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
+; CHECK-VS:       [[LOOP]]:
+; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-VS-NEXT:    [[GEP:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[IV]]
+; CHECK-VS-NEXT:    [[LOAD:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-VS-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-VS-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-VS-NEXT:    br i1 [[EC]], label %[[FOR_END]], label %[[LOOP]]
 ; CHECK-VS:       [[FOR_END]]:
-; CHECK-VS-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[TMP17]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP10]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[LOAD]], %[[LOOP]] ], [ [[TMP10]], %[[MIDDLE_BLOCK]] ], [ [[TMP17]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-VS-NEXT:    ret i32 [[LOAD_LCSSA]]
 ;
 entry:
@@ -282,7 +337,7 @@ define i32 @live_out_recurrence(ptr %A, i64 %n) {
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[SCALAR_RECUR_INIT:%.*]] = phi i32 [ [[VECTOR_RECUR_EXTRACT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -307,8 +362,18 @@ define i32 @live_out_recurrence(ptr %A, i64 %n) {
 ; CHECK-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
 ; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <8 x i32> [[TMP7]], i64 [[LAST_ACTIVE_LANE]]
 ; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[FOR:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[L:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    [[L]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[FOR_LCSSA:%.*]] = phi i32 [ [[TMP9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    [[FOR_LCSSA:%.*]] = phi i32 [ [[FOR]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ], [ [[TMP9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[FOR_LCSSA]]
 ;
 ; CHECK-VS-LABEL: define i32 @live_out_recurrence(
@@ -347,7 +412,7 @@ define i32 @live_out_recurrence(ptr %A, i64 %n) {
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-VS-NEXT:    [[SCALAR_RECUR_INIT:%.*]] = phi i32 [ [[VECTOR_RECUR_EXTRACT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -377,8 +442,18 @@ define i32 @live_out_recurrence(ptr %A, i64 %n) {
 ; CHECK-VS-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
 ; CHECK-VS-NEXT:    [[TMP23:%.*]] = extractelement <vscale x 8 x i32> [[TMP21]], i64 [[LAST_ACTIVE_LANE]]
 ; CHECK-VS-NEXT:    br label %[[EXIT]]
+; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
+; CHECK-VS:       [[LOOP]]:
+; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-VS-NEXT:    [[FOR:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[L:%.*]], %[[LOOP]] ]
+; CHECK-VS-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-VS-NEXT:    [[L]] = load i32, ptr [[GEP]], align 4
+; CHECK-VS-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-VS-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-VS-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK-VS:       [[EXIT]]:
-; CHECK-VS-NEXT:    [[FOR_LCSSA:%.*]] = phi i32 [ [[TMP23]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    [[FOR_LCSSA:%.*]] = phi i32 [ [[FOR]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ], [ [[TMP23]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-VS-NEXT:    ret i32 [[FOR_LCSSA]]
 ;
 entry:
@@ -441,7 +516,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP2]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <8 x i32> poison, i32 [[VAL]], i64 0
@@ -527,7 +602,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP2]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-VS-NEXT:    [[TMP21:%.*]] = call i32 @llvm.vscale.i32()
@@ -583,6 +658,140 @@ exit:
   ret void
 }
 
+define void @alloca(ptr %vla, i64 %N) {
+; CHECK-INVALID-COSTS-LABEL: Checking a loop in 'alloca'
+; CHECK-INVALID-COSTS: LV: epilogue tail-folding is enabled
+; CHECK-INVALID-COSTS: remark: <unknown>:0:0: This case of epilogue loop can't be tail-folded - Invalid costs
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ %iv.next, %for.body ], [ 0, %entry ]
+  %alloca = alloca i32, align 16
+  %arrayidx = getelementptr inbounds ptr, ptr %vla, i64 %iv
+  store ptr %alloca, ptr %arrayidx, align 8
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond.not = icmp eq i64 %iv.next, %N
+  br i1 %exitcond.not, label %for.end, label %for.body
+
+for.end:
+  call void @foo(ptr nonnull %vla)
+  ret void
+}
+declare void @foo(ptr)
+
+define i64 @find_last_offset_wide_canonical_iv(ptr %A, i64 %n) {
+; CHECK-INDUCTION-LABEL: define i64 @find_last_offset_wide_canonical_iv(
+; CHECK-INDUCTION-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-INDUCTION-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-INDUCTION-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
+; CHECK-INDUCTION-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-INDUCTION:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-INDUCTION-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 16
+; CHECK-INDUCTION-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-INDUCTION:       [[VECTOR_PH]]:
+; CHECK-INDUCTION-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 15
+; CHECK-INDUCTION-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-INDUCTION-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-INDUCTION:       [[VECTOR_BODY]]:
+; CHECK-INDUCTION-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-INDUCTION-NEXT:    [[VEC_IND:%.*]] = phi <8 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-INDUCTION-NEXT:    [[VEC_PHI:%.*]] = phi <8 x i64> [ splat (i64 -9223372036854775808), %[[VECTOR_PH]] ], [ [[TMP12:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-INDUCTION-NEXT:    [[VEC_PHI2:%.*]] = phi <8 x i64> [ splat (i64 -9223372036854775808), %[[VECTOR_PH]] ], [ [[TMP15:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-INDUCTION-NEXT:    [[STEP_ADD:%.*]] = add nuw <8 x i64> [[VEC_IND]], splat (i64 8)
+; CHECK-INDUCTION-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-INDUCTION-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 8
+; CHECK-INDUCTION-NEXT:    [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP1]], align 4
+; CHECK-INDUCTION-NEXT:    [[WIDE_LOAD3:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4
+; CHECK-INDUCTION-NEXT:    [[TMP3:%.*]] = icmp eq <8 x i32> [[WIDE_LOAD]], splat (i32 11)
+; CHECK-INDUCTION-NEXT:    [[TMP22:%.*]] = icmp eq <8 x i32> [[WIDE_LOAD3]], splat (i32 11)
+; CHECK-INDUCTION-NEXT:    [[TMP12]] = select <8 x i1> [[TMP3]], <8 x i64> [[VEC_IND]], <8 x i64> [[VEC_PHI]]
+; CHECK-INDUCTION-NEXT:    [[TMP15]] = select <8 x i1> [[TMP22]], <8 x i64> [[STEP_ADD]], <8 x i64> [[VEC_PHI2]]
+; CHECK-INDUCTION-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
+; CHECK-INDUCTION-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <8 x i64> [[STEP_ADD]], splat (i64 8)
+; CHECK-INDUCTION-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-INDUCTION-NEXT:    br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-INDUCTION:       [[MIDDLE_BLOCK]]:
+; CHECK-INDUCTION-NEXT:    [[RDX_MINMAX:%.*]] = call <8 x i64> @llvm.smax.v8i64(<8 x i64> [[TMP12]], <8 x i64> [[TMP15]])
+; CHECK-INDUCTION-NEXT:    [[TMP5:%.*]] = call i64 @llvm.vector.reduce.smax.v8i64(<8 x i64> [[RDX_MINMAX]])
+; CHECK-INDUCTION-NEXT:    [[TMP6:%.*]] = icmp ne i64 [[TMP5]], -9223372036854775808
+; CHECK-INDUCTION-NEXT:    [[TMP7:%.*]] = select i1 [[TMP6]], i64 [[TMP5]], i64 -1
+; CHECK-INDUCTION-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-INDUCTION-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-INDUCTION:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-INDUCTION-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
+; CHECK-INDUCTION:       [[VEC_EPILOG_PH]]:
+; CHECK-INDUCTION-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-INDUCTION-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ -1, %[[ITER_CHECK]] ], [ -1, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-INDUCTION-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[BC_MERGE_RDX]], -1
+; CHECK-INDUCTION-NEXT:    [[TMP9:%.*]] = select i1 [[TMP8]], i64 -9223372036854775808, i64 [[BC_MERGE_RDX]]
+; CHECK-INDUCTION-NEXT:    [[N_RND_UP:%.*]] = add i64 [[N]], 3
+; CHECK-INDUCTION-NEXT:    [[TMP10:%.*]] = and i64 [[N_RND_UP]], 3
+; CHECK-INDUCTION-NEXT:    [[N_VEC2:%.*]] = sub i64 [[N_RND_UP]], [[TMP10]]
+; CHECK-INDUCTION-NEXT:    [[TRIP_COUNT_MINUS_1:%.*]] = sub i64 [[N]], 1
+; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[TRIP_COUNT_MINUS_1]], i64 0
+; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLATINSERT5:%.*]] = insertelement <4 x i64> poison, i64 [[TMP9]], i64 0
+; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLAT6:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT5]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLATINSERT7:%.*]] = insertelement <4 x i64> poison, i64 [[VEC_EPILOG_RESUME_VAL]], i64 0
+; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLAT8:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT7]], <4 x i64> poison, <4 x i32> zeroinitializer
+; CHECK-INDUCTION-NEXT:    [[INDUCTION:%.*]] = add nuw <4 x i64> [[BROADCAST_SPLAT8]], <i64 0, i64 1, i64 2, i64 3>
+; CHECK-INDUCTION-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-INDUCTION:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-INDUCTION-NEXT:    [[TMP11:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[TMP17:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-INDUCTION-NEXT:    [[VEC_IND10:%.*]] = phi <4 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT14:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-INDUCTION-NEXT:    [[VEC_PHI11:%.*]] = phi <4 x i64> [ [[BROADCAST_SPLAT6]], %[[VEC_EPILOG_PH]] ], [ [[TMP23:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-INDUCTION-NEXT:    [[VEC_IND12:%.*]] = phi <4 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT15:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-INDUCTION-NEXT:    [[TMP14:%.*]] = icmp ule <4 x i64> [[VEC_IND12]], [[BROADCAST_SPLAT]]
+; CHECK-INDUCTION-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP11]]
+; CHECK-INDUCTION-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[TMP13]], <4 x i1> [[TMP14]], <4 x i32> poison)
+; CHECK-INDUCTION-NEXT:    [[TMP16:%.*]] = icmp eq <4 x i32> [[WIDE_MASKED_LOAD]], splat (i32 11)
+; CHECK-INDUCTION-NEXT:    [[TMP24:%.*]] = select <4 x i1> [[TMP14]], <4 x i1> [[TMP16]], <4 x i1> zeroinitializer
+; CHECK-INDUCTION-NEXT:    [[TMP23]] = select <4 x i1> [[TMP24]], <4 x i64> [[VEC_IND10]], <4 x i64> [[VEC_PHI11]]
+; CHECK-INDUCTION-NEXT:    [[TMP17]] = add i64 [[TMP11]], 4
+; CHECK-INDUCTION-NEXT:    [[VEC_IND_NEXT14]] = add nuw nsw <4 x i64> [[VEC_IND10]], splat (i64 4)
+; CHECK-INDUCTION-NEXT:    [[VEC_IND_NEXT15]] = add nuw <4 x i64> [[VEC_IND12]], splat (i64 4)
+; CHECK-INDUCTION-NEXT:    [[TMP18:%.*]] = icmp eq i64 [[TMP17]], [[N_VEC2]]
+; CHECK-INDUCTION-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-INDUCTION:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-INDUCTION-NEXT:    [[TMP19:%.*]] = call i64 @llvm.vector.reduce.smax.v4i64(<4 x i64> [[TMP23]])
+; CHECK-INDUCTION-NEXT:    [[TMP20:%.*]] = icmp ne i64 [[TMP19]], -9223372036854775808
+; CHECK-INDUCTION-NEXT:    [[TMP21:%.*]] = select i1 [[TMP20]], i64 [[TMP19]], i64 -1
+; CHECK-INDUCTION-NEXT:    br label %[[EXIT]]
+; CHECK-INDUCTION:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-INDUCTION-NEXT:    br label %[[LOOP:.*]]
+; CHECK-INDUCTION:       [[LOOP]]:
+; CHECK-INDUCTION-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-INDUCTION-NEXT:    [[RED:%.*]] = phi i64 [ -1, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SEL:%.*]], %[[LOOP]] ]
+; CHECK-INDUCTION-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-INDUCTION-NEXT:    [[L:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-INDUCTION-NEXT:    [[C:%.*]] = icmp eq i32 [[L]], 11
+; CHECK-INDUCTION-NEXT:    [[SEL]] = select i1 [[C]], i64 [[IV]], i64 [[RED]]
+; CHECK-INDUCTION-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-INDUCTION-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-INDUCTION-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
+; CHECK-INDUCTION:       [[EXIT]]:
+; CHECK-INDUCTION-NEXT:    [[SEL_LCSSA:%.*]] = phi i64 [ [[SEL]], %[[LOOP]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[TMP21]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-INDUCTION-NEXT:    ret i64 [[SEL_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %red = phi i64 [ -1, %entry ], [ %sel, %loop ]
+  %gep = getelementptr inbounds i32, ptr %A, i64 %iv
+  %l = load i32, ptr %gep, align 4
+  %c = icmp eq i32 %l, 11
+  %sel = select i1 %c, i64 %iv, i64 %red
+  %iv.next = add nuw nsw i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, %n
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  ret i64 %sel
+}
+
 define i64 @test_no_masked_interleave_support(i64 %y, i32 %n) {
 ; CHECK-INVALIDATE-INTERLEAVE-LABEL: Checking a loop in 'test_no_masked_interleave_support'
 ; CHECK-INVALIDATE-INTERLEAVE: LV: epilogue tail-folding is enabled
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll
index de28bb59b6bab..8196fcdc6dc98 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll
@@ -115,14 +115,14 @@ define i32 @pred_reduction(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE5]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP10]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE5]], [[PARTIAL_REDUCE]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP12]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -146,12 +146,32 @@ define i32 @pred_reduction(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT11]], i64 [[N]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = xor i1 [[TMP20]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE10]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_1:%.*]], %[[FOR_INC]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VAL:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CONV:%.*]] = zext i8 [[VAL]] to i32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM]], [[CONV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1]] = phi i32 [ [[ADD]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[SUM_1]], %[[FOR_INC]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ], [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
 ;
 entry:
@@ -292,14 +312,14 @@ define i32 @pred_reduction_sext(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE5]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP10]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE5]], [[PARTIAL_REDUCE]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP12]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -323,12 +343,32 @@ define i32 @pred_reduction_sext(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT11]], i64 [[N]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = xor i1 [[TMP20]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE10]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_1:%.*]], %[[FOR_INC]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VAL:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CONV:%.*]] = sext i8 [[VAL]] to i32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM]], [[CONV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1]] = phi i32 [ [[ADD]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[SUM_1]], %[[FOR_INC]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ], [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
 ;
 entry:
@@ -489,14 +529,14 @@ define i32 @pred_reduction_dotprod(ptr %a, ptr %b, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE7]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP16]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE7]], [[PARTIAL_REDUCE]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP18]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -524,12 +564,36 @@ define i32 @pred_reduction_dotprod(ptr %a, ptr %b, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT14]], i64 [[N]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP30:%.*]] = xor i1 [[TMP29]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP31:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE13]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_1:%.*]], %[[FOR_INC]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[LOAD_A:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX4:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[LOAD_B:%.*]] = load i8, ptr [[ARRAYIDX4]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXT_A:%.*]] = zext i8 [[LOAD_A]] to i32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXT_B:%.*]] = zext i8 [[LOAD_B]] to i32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MUL:%.*]] = mul nuw nsw i32 [[EXT_A]], [[EXT_B]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM]], [[MUL]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1]] = phi i32 [ [[ADD]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[TMP31]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP18]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[SUM_1]], %[[FOR_INC]] ], [ [[TMP18]], %[[MIDDLE_BLOCK]] ], [ [[TMP31]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
 ;
 entry:
@@ -696,7 +760,7 @@ define i32 @pred_sub_reduction(ptr %a, ptr %b, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE7]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP16]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE7]], [[PARTIAL_REDUCE]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
@@ -704,7 +768,7 @@ define i32 @pred_sub_reduction(ptr %a, ptr %b, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP19]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -731,13 +795,37 @@ define i32 @pred_sub_reduction(ptr %a, ptr %b, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT14]], i64 [[N]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP30:%.*]] = xor i1 [[TMP29]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP31:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE13]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP32:%.*]] = sub i32 [[BC_MERGE_RDX]], [[TMP31]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_1:%.*]], %[[FOR_INC]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[LOAD_A:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX4:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[LOAD_B:%.*]] = load i8, ptr [[ARRAYIDX4]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXT_A:%.*]] = zext i8 [[LOAD_A]] to i32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXT_B:%.*]] = zext i8 [[LOAD_B]] to i32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MUL:%.*]] = mul nuw nsw i32 [[EXT_A]], [[EXT_B]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUB:%.*]] = sub nsw i32 [[SUM]], [[MUL]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1]] = phi i32 [ [[SUB]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[TMP32]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP19]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[SUM_1]], %[[FOR_INC]] ], [ [[TMP19]], %[[MIDDLE_BLOCK]] ], [ [[TMP32]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
 ;
 entry:
@@ -903,14 +991,14 @@ define i32 @chained_pred_reduction(ptr %src, ptr noalias %src_b, ptr %cond, i64
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE9]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[PARTIAL_REDUCE5]], <16 x i32> [[TMP14]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE9]], [[PARTIAL_REDUCE8]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP16]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -939,12 +1027,36 @@ define i32 @chained_pred_reduction(ptr %src, ptr noalias %src_b, ptr %cond, i64
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT17]], i64 [[N]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP27:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP28:%.*]] = xor i1 [[TMP27]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP28]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP28]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE16]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_2:%.*]], %[[FOR_INC]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VAL:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CONV:%.*]] = zext i8 [[VAL]] to i32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM]], [[CONV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1:%.*]] = phi i32 [ [[ADD]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[B_GEP:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BVAL:%.*]] = load i8, ptr [[B_GEP]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BCONV:%.*]] = zext i8 [[BVAL]] to i32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2]] = add nsw i32 [[SUM_1]], [[BCONV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2_LCSSA:%.*]] = phi i32 [ [[TMP29]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP16]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2_LCSSA:%.*]] = phi i32 [ [[SUM_2]], %[[FOR_INC]] ], [ [[TMP16]], %[[MIDDLE_BLOCK]] ], [ [[TMP29]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_2_LCSSA]]
 ;
 entry:
@@ -1110,14 +1222,14 @@ define i32 @reduction_before_pred(ptr %src, ptr noalias %src_b, ptr %cond, i64 %
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE9]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[PARTIAL_REDUCE4]], <16 x i32> [[TMP14]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE9]], [[PARTIAL_REDUCE8]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP16]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -1146,12 +1258,36 @@ define i32 @reduction_before_pred(ptr %src, ptr noalias %src_b, ptr %cond, i64 %
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT17]], i64 [[N]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP27:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP28:%.*]] = xor i1 [[TMP27]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP28]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP28]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE16]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_2:%.*]], %[[FOR_INC]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[B_GEP:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BVAL:%.*]] = load i8, ptr [[B_GEP]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BCONV:%.*]] = zext i8 [[BVAL]] to i32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1:%.*]] = add nsw i32 [[SUM]], [[BCONV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VAL:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CONV:%.*]] = zext i8 [[VAL]] to i32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM_1]], [[CONV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2]] = phi i32 [ [[SUM_1]], %[[FOR_BODY]] ], [ [[ADD]], %[[IF_THEN]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2_LCSSA:%.*]] = phi i32 [ [[TMP29]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP16]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2_LCSSA:%.*]] = phi i32 [ [[SUM_2]], %[[FOR_INC]] ], [ [[TMP16]], %[[MIDDLE_BLOCK]] ], [ [[TMP29]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_2_LCSSA]]
 ;
 entry:
@@ -1263,7 +1399,7 @@ define i32 @pred_reduction_incoming_1(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 16 x i1> @llvm.get.active.lane.mask.nxv16i1.i64(i64 [[INDEX_NEXT]], i64 [[N]])
 ; CHECK-TAILFOLD-NEXT:    [[TMP12:%.*]] = extractelement <vscale x 16 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-TAILFOLD-NEXT:    [[TMP13:%.*]] = xor i1 [[TMP12]], true
-; CHECK-TAILFOLD-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK-TAILFOLD-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-TAILFOLD:       [[MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-NEXT:    [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.nxv4i32(<vscale x 4 x i32> [[PARTIAL_REDUCE]])
 ; CHECK-TAILFOLD-NEXT:    br label %[[EXIT:.*]]
@@ -1304,14 +1440,14 @@ define i32 @pred_reduction_incoming_1(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE5]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP10]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP14:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE5]], [[PARTIAL_REDUCE]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP12]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -1335,12 +1471,32 @@ define i32 @pred_reduction_incoming_1(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT11]], i64 [[N]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = xor i1 [[TMP20]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE10]])
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN:.*]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC]], i64 [[IV:%.*]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VAL:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CONV:%.*]] = zext i8 [[VAL]] to i32
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM:%.*]], [[CONV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC:.*]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_1:%.*]], %[[FOR_INC]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN]]
+; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1]] = phi i32 [ [[ADD]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
 ; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[SUM_1]], %[[FOR_INC]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ], [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
 ;
 entry:
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll
index b2d5b6a3322ea..e1086979859cf 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll
@@ -94,7 +94,7 @@ define void @dead_for(ptr %a, i64 %N) {
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_PH]]:
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_RND_UP:%.*]] = add i64 [[N]], 3
@@ -121,6 +121,18 @@ define void @dead_for(ptr %a, i64 %N) {
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[TMP9]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[LOOP:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[LOOP]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR:%.*]] = phi i64 [ 99, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[L:%.*]], %[[LOOP]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[GEP:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L]] = load i64, ptr [[GEP]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[ADD:%.*]] = add i64 [[L]], 10
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store i64 [[ADD]], ptr [[GEP]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK-TAILFOLDED-EPILOGUE:       [[EXIT]]:
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    ret void
 ;
@@ -346,7 +358,7 @@ define i64 @for_phi_not_used_in_loop_and_live_out(ptr %a, i64 %N) {
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_PH]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_PH]]:
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[SCALAR_RECUR_INIT:%.*]] = phi i64 [ [[VECTOR_RECUR_EXTRACT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 99, %[[ITER_CHECK]] ], [ 99, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
@@ -382,9 +394,21 @@ define i64 @for_phi_not_used_in_loop_and_live_out(ptr %a, i64 %N) {
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 [[LAST_ACTIVE_LANE]]
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[EXIT]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[LOOP:.*]]
+; CHECK-TAILFOLDED-EPILOGUE:       [[LOOP]]:
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR:%.*]] = phi i64 [ 99, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[L:%.*]], %[[LOOP]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[GEP:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L]] = load i64, ptr [[GEP]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[ADD:%.*]] = add i64 [[L]], 10
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store i64 [[ADD]], ptr [[GEP]], align 4
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK-TAILFOLDED-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR_LCSSA:%.*]] = phi i64 [ [[TMP11]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L_LCSSA:%.*]] = phi i64 [ [[TMP12]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR_LCSSA:%.*]] = phi i64 [ [[FOR]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ], [ [[TMP11]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L_LCSSA:%.*]] = phi i64 [ [[L]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ], [ [[TMP12]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[RES:%.*]] = add i64 [[L_LCSSA]], [[FOR_LCSSA]]
 ; CHECK-TAILFOLDED-EPILOGUE-NEXT:    ret i64 [[RES]]
 ;
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index d77ce077c06f0..70878c60811ca 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -36,10 +36,10 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i8 %val) {
 ; CHECK-DISABLED-EPILOG: remark: <unknown>:0:0: Options conflict, epilogue vectorization is disallowed while epilogue tail-folding allowed!
 ;
 ; CHECK-NO-FORCED-MAIN-VF-LABEL: Checking a loop in 'test_epilogue_tf'
-; CHECK-NO-FORCED-MAIN-VF: remark: <unknown>:0:0: For now, epilogue tail-folding can't be applied without forced main/epilogue loop VF
+; CHECK-NO-FORCED-MAIN-VF-NOT: LV: epilogue tail-folding is enabled
 
 ; CHECK-NO-FORCED-EPILOGUE-VF-LABEL: Checking a loop in 'test_epilogue_tf'
-; CHECK-NO-FORCED-EPILOGUE-VF: remark: <unknown>:0:0: For now, epilogue tail-folding can't be applied without forced main/epilogue loop VF
+; CHECK-NO-FORCED-EPILOGUE-VF-NOT: LV: epilogue tail-folding is enabled
 ;
 ; CHECK-INVALID-VFs-LABEL: Checking a loop in 'test_epilogue_tf'
 ; CHECK-INVALID-VFs: remark: <unknown>:0:0: For now, epilogue tail-folding can't be applied when VF of the main loop <= VF of the epilogue
@@ -198,26 +198,31 @@ for.end:
   ret i32 %result
 }
 
-define i1 @early_exit(ptr %A, i64 %n, i8 %find) {
+define i64 @early_exit(ptr dereferenceable(1024) align 8 %src, i1 %cond) {
 ; CHECK-DISABLED-EARLY-EXIT-LABEL: LV: Checking a loop in 'early_exit'
 ; CHECK-DISABLED-EARLY-EXIT: remark: <unknown>:0:0: Epilogue tail-folding is not supported yet for early-exit loops
 ;
 entry:
-  br label %for.body
+  br label %loop.header
 
-for.body:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %cont ]
-  %arrayidx = getelementptr inbounds i8, ptr %A, i64 %iv
-  %val = load i8, ptr %arrayidx, align 1
-  %exitcond = icmp eq i8 %val, %find
-  br i1 %exitcond, label %exit, label %cont
+loop.header:
+  %iv = phi i64 [ %iv.next, %latch ], [ 0, %entry ]
+  %gep = getelementptr inbounds double, ptr %src, i64 %iv
+  %val = load double, ptr %gep, align 8
+  %neg = fneg double %val
+  %c.1 = fcmp une double %neg, 10.0
+  br i1 %c.1, label %latch, label %early.exit
+
+latch:
+  %iv.next = add nuw i64 %iv, 1
+  %exit.cond = icmp eq i64 %iv.next, 127
+  br i1 %exit.cond, label %exit, label %loop.header
+
+early.exit:
+  ret i64 %iv
 
-cont:
-  %iv.next = add nuw nsw i64 %iv, 1
-  %contcond = icmp ne i64 %iv.next, %n
-  br i1 %contcond, label %for.body, label %exit
 exit:
-  ret i1 %exitcond
+  ret i64 10
 }
 
 ; For this function, the check line is not related to epilogue tail-folding, but when vectorizing this case gets supported,
@@ -280,7 +285,6 @@ for.end:
 
 define void @test_outer_loop(ptr %A, i64 %m) {
 ; CHECK-OUTER-LOOP-LABEL: Checking a loop in 'test_outer_loop'
-; CHECK-OUTER-LOOP: remark: <unknown>:0:0: Epilogue tail-folding is not supported for outer loop
 ; CHECK-OUTER-LOOP-NOT: LV: epilogue tail-folding is enabled
 ;
 entry:

>From 6f51e6a35f116abb0942377d856672fdb86028aa Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Tue, 15 Sep 2026 12:58:46 +0100
Subject: [PATCH 17/25] Make IAI member of Planner to keep it alive after
 planForEpilogueTF()  as it's referenced by Interleave Recipes

---
 .../Vectorize/LoopVectorizationPlanner.h      |   2 +
 .../Transforms/Vectorize/LoopVectorize.cpp    |  12 +-
 .../LoopVectorize/AArch64/sve-widen-phi.ll    | 274 ++++++++++++++++++
 3 files changed, 282 insertions(+), 6 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index d870f049162db..847a9026aa307 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -893,6 +893,8 @@ class LoopVectorizationPlanner {
 
   /// The interleaved access analysis.
   InterleavedAccessInfo &IAI;
+  /// The interleaved access analysis for the case of tail-folded epilogue.
+  std::unique_ptr<InterleavedAccessInfo> EpilogueTfIAI;
 
   PredicatedScalarEvolution &PSE;
 
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index af50d338856a4..e0012aa9628a9 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3022,7 +3022,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
   // TODO: Make NoScalarEpilogueNeeded lambda a separate function to be used
   // only for main loop VF not also epilogueVF. Using it for epilogueVF against
   // full TC is inaccurate.
-  auto NoScalarEpilogueNeeded = [this, &UserIC](unsigned MaxRuntimeVF) {
+  auto NoScalarEpilogueNeeded = [this, &UserIC](uint64_t MaxRuntimeVF) {
     // Return false if the loop is neither a single-latch-exit loop nor an
     // early-exit loop as tail-folding is not supported in that case.
     if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
@@ -5477,13 +5477,13 @@ bool LoopVectorizationPlanner::planForEpilogueTF() {
   if (EnableInterleavedMemAccesses.getNumOccurrences() > 0)
     UseInterleaved = EnableInterleavedMemAccesses;
 
-  InterleavedAccessInfo EpilogueTfCMIAI(PSE, OrigLoop, DT, LI, Legal->getLAI(),
-                                        Config.OptForSize);
+  EpilogueTfIAI = std::make_unique<InterleavedAccessInfo>(
+      PSE, OrigLoop, DT, LI, Legal->getLAI(), Config.OptForSize);
   if (UseInterleaved)
-    EpilogueTfCMIAI.analyzeInterleaving(useMaskedInterleavedAccesses(TTI));
+    EpilogueTfIAI->analyzeInterleaving(useMaskedInterleavedAccesses(TTI));
   LoopVectorizationCostModel EpilogueTfCM(
       EpilogueTailLoweringStatus, OrigLoop, PSE, LI, Legal, TTI, TLI, CM->AC,
-      ORE, CM->GetBFI, CM->TheFunction, EpilogueTfCMIAI, Config);
+      ORE, CM->GetBFI, CM->TheFunction, *EpilogueTfIAI, Config);
 
   assert(EpilogueTfCM.preferTailFoldedLoop() &&
          "Epilogue tail-folding is expected to be enabled");
@@ -8282,7 +8282,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Destroy the cost model before executing any plan, so that code generation
   // cannot rely on cost-modeling decisions.
-  LVP.clearCostModel();
+  // LVP.clearCostModel();
 
   VPlan &BestPlan = *BestPlanPtr;
   // Consider vectorizing the epilogue too if it's profitable.
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
index 6240954578dee..1f36d26c59fda 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
@@ -2,6 +2,12 @@
 ; RUN: opt -mtriple aarch64-linux-gnu -mattr=+sve -passes=loop-vectorize -S \
 ; RUN:   -tail-folding-policy=dont-fold-tail < %s | FileCheck %s
 
+; RUN: opt -S -p loop-vectorize -mattr=+sve -force-vector-width="vscale x 4" \
+; RUN: -epilogue-vectorization-force-VF="vscale x 2" \
+; RUN: -epilogue-tail-folding-policy=prefer-fold-tail %s | FileCheck %s --check-prefix=CHECK-EPI-TF
+
+target triple = "aarch64-unknown-linux-gnu"
+
 ; Ensure that we can vectorize loops such as:
 ;   int *ptr = c;
 ;   for (long long i = 0; i < n; i++) {
@@ -84,6 +90,109 @@ define void @widen_ptr_phi_unrolled(ptr noalias nocapture %a, ptr noalias nocapt
 ; CHECK:       for.exit:
 ; CHECK-NEXT:    ret void
 ;
+; CHECK-EPI-TF-LABEL: @widen_ptr_phi_unrolled(
+; CHECK-EPI-TF-NEXT:  iter.check:
+; CHECK-EPI-TF-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-EPI-TF-NEXT:    [[TMP1:%.*]] = shl nuw nsw i64 [[TMP0]], 1
+; CHECK-EPI-TF-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N:%.*]], [[TMP1]]
+; CHECK-EPI-TF-NEXT:    br i1 [[MIN_ITERS_CHECK]], label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_MAIN_LOOP_ITER_CHECK:%.*]]
+; CHECK-EPI-TF:       vector.main.loop.iter.check:
+; CHECK-EPI-TF-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-EPI-TF-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
+; CHECK-EPI-TF-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label [[VEC_EPILOG_PH]], label [[VECTOR_PH:%.*]]
+; CHECK-EPI-TF:       vector.ph:
+; CHECK-EPI-TF-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-EPI-TF-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; CHECK-EPI-TF-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-EPI-TF-NEXT:    [[TMP4:%.*]] = shl i64 [[N_VEC]], 3
+; CHECK-EPI-TF-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[C:%.*]], i64 [[TMP4]]
+; CHECK-EPI-TF-NEXT:    br label [[VECTOR_BODY:%.*]]
+; CHECK-EPI-TF:       vector.body:
+; CHECK-EPI-TF-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP6:%.*]] = shl i64 [[INDEX]], 3
+; CHECK-EPI-TF-NEXT:    [[TMP7:%.*]] = add i64 [[TMP3]], 0
+; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = mul i64 [[TMP7]], 8
+; CHECK-EPI-TF-NEXT:    [[TMP9:%.*]] = add i64 [[TMP6]], [[TMP8]]
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP6]]
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP2:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP9]]
+; CHECK-EPI-TF-NEXT:    [[WIDE_VEC:%.*]] = load <vscale x 8 x i32>, ptr [[NEXT_GEP]], align 4
+; CHECK-EPI-TF-NEXT:    [[STRIDED_VEC:%.*]] = call { <vscale x 4 x i32>, <vscale x 4 x i32> } @llvm.vector.deinterleave2.nxv8i32(<vscale x 8 x i32> [[WIDE_VEC]])
+; CHECK-EPI-TF-NEXT:    [[TMP10:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 0
+; CHECK-EPI-TF-NEXT:    [[TMP11:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 1
+; CHECK-EPI-TF-NEXT:    [[WIDE_VEC3:%.*]] = load <vscale x 8 x i32>, ptr [[NEXT_GEP2]], align 4
+; CHECK-EPI-TF-NEXT:    [[STRIDED_VEC4:%.*]] = call { <vscale x 4 x i32>, <vscale x 4 x i32> } @llvm.vector.deinterleave2.nxv8i32(<vscale x 8 x i32> [[WIDE_VEC3]])
+; CHECK-EPI-TF-NEXT:    [[TMP12:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC4]], 0
+; CHECK-EPI-TF-NEXT:    [[TMP13:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC4]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP14:%.*]] = add nsw <vscale x 4 x i32> [[TMP10]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP15:%.*]] = add nsw <vscale x 4 x i32> [[TMP12]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[A:%.*]], i64 [[INDEX]]
+; CHECK-EPI-TF-NEXT:    [[TMP17:%.*]] = getelementptr inbounds i32, ptr [[TMP16]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP14]], ptr [[TMP16]], align 4
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP15]], ptr [[TMP17]], align 4
+; CHECK-EPI-TF-NEXT:    [[TMP18:%.*]] = add nsw <vscale x 4 x i32> [[TMP11]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP19:%.*]] = add nsw <vscale x 4 x i32> [[TMP13]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[B:%.*]], i64 [[INDEX]]
+; CHECK-EPI-TF-NEXT:    [[TMP21:%.*]] = getelementptr inbounds i32, ptr [[TMP20]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP18]], ptr [[TMP20]], align 4
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP19]], ptr [[TMP21]], align 4
+; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-EPI-TF-NEXT:    [[TMP22:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP22]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-EPI-TF:       middle.block:
+; CHECK-EPI-TF-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-EPI-TF-NEXT:    br i1 [[CMP_N]], label [[FOR_EXIT:%.*]], label [[VEC_EPILOG_ITER_CHECK:%.*]]
+; CHECK-EPI-TF:       vec.epilog.iter.check:
+; CHECK-EPI-TF-NEXT:    br i1 false, label [[VEC_EPILOG_SCALAR_PH:%.*]], label [[VEC_EPILOG_PH]]
+; CHECK-EPI-TF:       vec.epilog.ph:
+; CHECK-EPI-TF-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[ITER_CHECK:%.*]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP23:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-EPI-TF-NEXT:    [[TMP24:%.*]] = shl nuw i64 [[TMP23]], 1
+; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-EPI-TF-NEXT:    br label [[VEC_EPILOG_VECTOR_BODY:%.*]]
+; CHECK-EPI-TF:       vec.epilog.vector.body:
+; CHECK-EPI-TF-NEXT:    [[INDEX5:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 2 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], [[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP25:%.*]] = shl i64 [[INDEX5]], 3
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP6:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP25]]
+; CHECK-EPI-TF-NEXT:    [[INTERLEAVED_MASK:%.*]] = call <vscale x 4 x i1> @llvm.vector.interleave2.nxv4i1(<vscale x 2 x i1> [[ACTIVE_LANE_MASK]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-EPI-TF-NEXT:    [[WIDE_MASKED_VEC:%.*]] = call <vscale x 4 x i32> @llvm.masked.load.nxv4i32.p0(ptr align 4 [[NEXT_GEP6]], <vscale x 4 x i1> [[INTERLEAVED_MASK]], <vscale x 4 x i32> poison)
+; CHECK-EPI-TF-NEXT:    [[STRIDED_VEC7:%.*]] = call { <vscale x 2 x i32>, <vscale x 2 x i32> } @llvm.vector.deinterleave2.nxv4i32(<vscale x 4 x i32> [[WIDE_MASKED_VEC]])
+; CHECK-EPI-TF-NEXT:    [[TMP26:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32> } [[STRIDED_VEC7]], 0
+; CHECK-EPI-TF-NEXT:    [[TMP27:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32> } [[STRIDED_VEC7]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP28:%.*]] = add nsw <vscale x 2 x i32> [[TMP26]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP29:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX5]]
+; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP28]], ptr align 4 [[TMP29]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-EPI-TF-NEXT:    [[TMP30:%.*]] = add nsw <vscale x 2 x i32> [[TMP27]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP31:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX5]]
+; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP30]], ptr align 4 [[TMP31]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT8]] = add i64 [[INDEX5]], [[TMP24]]
+; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[INDEX_NEXT8]], i64 [[N]])
+; CHECK-EPI-TF-NEXT:    [[TMP32:%.*]] = extractelement <vscale x 2 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-EPI-TF-NEXT:    [[TMP33:%.*]] = xor i1 [[TMP32]], true
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP33]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-EPI-TF:       vec.epilog.middle.block:
+; CHECK-EPI-TF-NEXT:    br label [[FOR_EXIT]]
+; CHECK-EPI-TF:       vec.epilog.scalar.ph:
+; CHECK-EPI-TF-NEXT:    br label [[FOR_BODY:%.*]]
+; CHECK-EPI-TF:       for.body:
+; CHECK-EPI-TF-NEXT:    [[PTR_014:%.*]] = phi ptr [ [[INCDEC_PTR1:%.*]], [[FOR_BODY]] ], [ [[C]], [[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-EPI-TF-NEXT:    [[I_013:%.*]] = phi i64 [ [[INC:%.*]], [[FOR_BODY]] ], [ 0, [[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-EPI-TF-NEXT:    [[INCDEC_PTR:%.*]] = getelementptr inbounds i32, ptr [[PTR_014]], i64 1
+; CHECK-EPI-TF-NEXT:    [[TMP34:%.*]] = load i32, ptr [[PTR_014]], align 4
+; CHECK-EPI-TF-NEXT:    [[INCDEC_PTR1]] = getelementptr inbounds i32, ptr [[PTR_014]], i64 2
+; CHECK-EPI-TF-NEXT:    [[TMP35:%.*]] = load i32, ptr [[INCDEC_PTR]], align 4
+; CHECK-EPI-TF-NEXT:    [[ADD:%.*]] = add nsw i32 [[TMP34]], 1
+; CHECK-EPI-TF-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[I_013]]
+; CHECK-EPI-TF-NEXT:    store i32 [[ADD]], ptr [[ARRAYIDX]], align 4
+; CHECK-EPI-TF-NEXT:    [[ADD2:%.*]] = add nsw i32 [[TMP35]], 1
+; CHECK-EPI-TF-NEXT:    [[ARRAYIDX3:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[I_013]]
+; CHECK-EPI-TF-NEXT:    store i32 [[ADD2]], ptr [[ARRAYIDX3]], align 4
+; CHECK-EPI-TF-NEXT:    [[INC]] = add nuw nsw i64 [[I_013]], 1
+; CHECK-EPI-TF-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INC]], [[N]]
+; CHECK-EPI-TF-NEXT:    br i1 [[EXITCOND_NOT]], label [[FOR_EXIT]], label [[FOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-EPI-TF:       for.exit:
+; CHECK-EPI-TF-NEXT:    ret void
+;
 entry:
   br label %for.body
 
@@ -174,6 +283,84 @@ define void @widen_2ptrs_phi_unrolled(ptr noalias nocapture %dst, ptr noalias no
 ; CHECK:       for.cond.cleanup:
 ; CHECK-NEXT:    ret void
 ;
+; CHECK-EPI-TF-LABEL: @widen_2ptrs_phi_unrolled(
+; CHECK-EPI-TF-NEXT:  iter.check:
+; CHECK-EPI-TF-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-EPI-TF-NEXT:    [[TMP1:%.*]] = shl nuw nsw i64 [[TMP0]], 1
+; CHECK-EPI-TF-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N:%.*]], [[TMP1]]
+; CHECK-EPI-TF-NEXT:    br i1 [[MIN_ITERS_CHECK]], label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_MAIN_LOOP_ITER_CHECK:%.*]]
+; CHECK-EPI-TF:       vector.main.loop.iter.check:
+; CHECK-EPI-TF-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-EPI-TF-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
+; CHECK-EPI-TF-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label [[VEC_EPILOG_PH]], label [[VECTOR_PH:%.*]]
+; CHECK-EPI-TF:       vector.ph:
+; CHECK-EPI-TF-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-EPI-TF-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; CHECK-EPI-TF-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-EPI-TF-NEXT:    [[TMP4:%.*]] = shl i64 [[N_VEC]], 2
+; CHECK-EPI-TF-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[SRC:%.*]], i64 [[TMP4]]
+; CHECK-EPI-TF-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[DST:%.*]], i64 [[TMP4]]
+; CHECK-EPI-TF-NEXT:    br label [[VECTOR_BODY:%.*]]
+; CHECK-EPI-TF:       vector.body:
+; CHECK-EPI-TF-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP7:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP7]]
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP2:%.*]] = getelementptr i8, ptr [[DST]], i64 [[TMP7]]
+; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = getelementptr i32, ptr [[NEXT_GEP]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[NEXT_GEP]], align 4
+; CHECK-EPI-TF-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 4 x i32>, ptr [[TMP8]], align 4
+; CHECK-EPI-TF-NEXT:    [[TMP9:%.*]] = shl nsw <vscale x 4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP10:%.*]] = shl nsw <vscale x 4 x i32> [[WIDE_LOAD3]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP11:%.*]] = getelementptr i32, ptr [[NEXT_GEP2]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP9]], ptr [[NEXT_GEP2]], align 4
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP10]], ptr [[TMP11]], align 4
+; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-EPI-TF-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP12]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-EPI-TF:       middle.block:
+; CHECK-EPI-TF-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-EPI-TF-NEXT:    br i1 [[CMP_N]], label [[FOR_COND_CLEANUP:%.*]], label [[VEC_EPILOG_ITER_CHECK:%.*]]
+; CHECK-EPI-TF:       vec.epilog.iter.check:
+; CHECK-EPI-TF-NEXT:    br i1 false, label [[VEC_EPILOG_SCALAR_PH:%.*]], label [[VEC_EPILOG_PH]]
+; CHECK-EPI-TF:       vec.epilog.ph:
+; CHECK-EPI-TF-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[ITER_CHECK:%.*]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP13:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-EPI-TF-NEXT:    [[TMP14:%.*]] = shl nuw i64 [[TMP13]], 1
+; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-EPI-TF-NEXT:    br label [[VEC_EPILOG_VECTOR_BODY:%.*]]
+; CHECK-EPI-TF:       vec.epilog.vector.body:
+; CHECK-EPI-TF-NEXT:    [[INDEX5:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 2 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], [[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP15:%.*]] = shl i64 [[INDEX5]], 2
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP6:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP15]]
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP7:%.*]] = getelementptr i8, ptr [[DST]], i64 [[TMP15]]
+; CHECK-EPI-TF-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 2 x i32> @llvm.masked.load.nxv2i32.p0(ptr align 4 [[NEXT_GEP6]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]], <vscale x 2 x i32> poison)
+; CHECK-EPI-TF-NEXT:    [[TMP16:%.*]] = shl nsw <vscale x 2 x i32> [[WIDE_MASKED_LOAD]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP16]], ptr align 4 [[NEXT_GEP7]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT8]] = add i64 [[INDEX5]], [[TMP14]]
+; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[INDEX_NEXT8]], i64 [[N]])
+; CHECK-EPI-TF-NEXT:    [[TMP17:%.*]] = extractelement <vscale x 2 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-EPI-TF-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP18]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-EPI-TF:       vec.epilog.middle.block:
+; CHECK-EPI-TF-NEXT:    br label [[FOR_COND_CLEANUP]]
+; CHECK-EPI-TF:       vec.epilog.scalar.ph:
+; CHECK-EPI-TF-NEXT:    br label [[FOR_BODY:%.*]]
+; CHECK-EPI-TF:       for.body:
+; CHECK-EPI-TF-NEXT:    [[I_011:%.*]] = phi i64 [ [[INC:%.*]], [[FOR_BODY]] ], [ 0, [[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-EPI-TF-NEXT:    [[S_010:%.*]] = phi ptr [ [[INCDEC_PTR1:%.*]], [[FOR_BODY]] ], [ [[SRC]], [[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-EPI-TF-NEXT:    [[D_09:%.*]] = phi ptr [ [[INCDEC_PTR:%.*]], [[FOR_BODY]] ], [ [[DST]], [[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP19:%.*]] = load i32, ptr [[S_010]], align 4
+; CHECK-EPI-TF-NEXT:    [[MUL:%.*]] = shl nsw i32 [[TMP19]], 1
+; CHECK-EPI-TF-NEXT:    store i32 [[MUL]], ptr [[D_09]], align 4
+; CHECK-EPI-TF-NEXT:    [[INCDEC_PTR]] = getelementptr inbounds i32, ptr [[D_09]], i64 1
+; CHECK-EPI-TF-NEXT:    [[INCDEC_PTR1]] = getelementptr inbounds i32, ptr [[S_010]], i64 1
+; CHECK-EPI-TF-NEXT:    [[INC]] = add nuw nsw i64 [[I_011]], 1
+; CHECK-EPI-TF-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INC]], [[N]]
+; CHECK-EPI-TF-NEXT:    br i1 [[EXITCOND_NOT]], label [[FOR_COND_CLEANUP]], label [[FOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK-EPI-TF:       for.cond.cleanup:
+; CHECK-EPI-TF-NEXT:    ret void
+;
 entry:
   br label %for.body
 
@@ -263,6 +450,66 @@ define i32 @pointer_iv_mixed(ptr noalias %a, ptr noalias %b, i64 %n) #0 {
 ; CHECK-NEXT:    [[VAR5:%.*]] = phi i32 [ [[VAR2]], [[FOR_BODY]] ], [ [[TMP14]], [[MIDDLE_BLOCK]] ]
 ; CHECK-NEXT:    ret i32 [[VAR5]]
 ;
+; CHECK-EPI-TF-LABEL: @pointer_iv_mixed(
+; CHECK-EPI-TF-NEXT:  entry:
+; CHECK-EPI-TF-NEXT:    [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N:%.*]], i64 1)
+; CHECK-EPI-TF-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-EPI-TF-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 1
+; CHECK-EPI-TF-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; CHECK-EPI-TF-NEXT:    br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
+; CHECK-EPI-TF:       vector.ph:
+; CHECK-EPI-TF-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
+; CHECK-EPI-TF-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
+; CHECK-EPI-TF-NEXT:    [[TMP3:%.*]] = shl i64 [[N_VEC]], 2
+; CHECK-EPI-TF-NEXT:    [[TMP4:%.*]] = getelementptr i8, ptr [[A:%.*]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    [[TMP5:%.*]] = shl i64 [[N_VEC]], 3
+; CHECK-EPI-TF-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[B:%.*]], i64 [[TMP5]]
+; CHECK-EPI-TF-NEXT:    br label [[VECTOR_BODY:%.*]]
+; CHECK-EPI-TF:       vector.body:
+; CHECK-EPI-TF-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[POINTER_PHI:%.*]] = phi ptr [ [[A]], [[VECTOR_PH]] ], [ [[PTR_IND:%.*]], [[VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 2 x i32> [ zeroinitializer, [[VECTOR_PH]] ], [ [[TMP11:%.*]], [[VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP7:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = shl <vscale x 2 x i64> [[TMP7]], splat (i64 2)
+; CHECK-EPI-TF-NEXT:    [[VECTOR_GEP:%.*]] = getelementptr i8, ptr [[POINTER_PHI]], <vscale x 2 x i64> [[TMP8]]
+; CHECK-EPI-TF-NEXT:    [[TMP9:%.*]] = extractelement <vscale x 2 x ptr> [[VECTOR_GEP]], i64 0
+; CHECK-EPI-TF-NEXT:    [[TMP10:%.*]] = shl i64 [[INDEX]], 3
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP10]]
+; CHECK-EPI-TF-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 2 x i32>, ptr [[TMP9]], align 8
+; CHECK-EPI-TF-NEXT:    [[TMP11]] = add <vscale x 2 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
+; CHECK-EPI-TF-NEXT:    store <vscale x 2 x ptr> [[VECTOR_GEP]], ptr [[NEXT_GEP]], align 8
+; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-EPI-TF-NEXT:    [[TMP12:%.*]] = shl i64 [[TMP2]], 2
+; CHECK-EPI-TF-NEXT:    [[PTR_IND]] = getelementptr i8, ptr [[POINTER_PHI]], i64 [[TMP12]]
+; CHECK-EPI-TF-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP13]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; CHECK-EPI-TF:       middle.block:
+; CHECK-EPI-TF-NEXT:    [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.nxv2i32(<vscale x 2 x i32> [[TMP11]])
+; CHECK-EPI-TF-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-EPI-TF-NEXT:    br i1 [[CMP_N]], label [[FOR_END:%.*]], label [[SCALAR_PH]]
+; CHECK-EPI-TF:       scalar.ph:
+; CHECK-EPI-TF-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[MIDDLE_BLOCK]] ], [ 0, [[ENTRY:%.*]] ]
+; CHECK-EPI-TF-NEXT:    [[BC_RESUME_VAL1:%.*]] = phi ptr [ [[TMP4]], [[MIDDLE_BLOCK]] ], [ [[A]], [[ENTRY]] ]
+; CHECK-EPI-TF-NEXT:    [[BC_RESUME_VAL2:%.*]] = phi ptr [ [[TMP6]], [[MIDDLE_BLOCK]] ], [ [[B]], [[ENTRY]] ]
+; CHECK-EPI-TF-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP14]], [[MIDDLE_BLOCK]] ], [ 0, [[ENTRY]] ]
+; CHECK-EPI-TF-NEXT:    br label [[FOR_BODY:%.*]]
+; CHECK-EPI-TF:       for.body:
+; CHECK-EPI-TF-NEXT:    [[I:%.*]] = phi i64 [ [[I_NEXT:%.*]], [[FOR_BODY]] ], [ [[BC_RESUME_VAL]], [[SCALAR_PH]] ]
+; CHECK-EPI-TF-NEXT:    [[P:%.*]] = phi ptr [ [[VAR3:%.*]], [[FOR_BODY]] ], [ [[BC_RESUME_VAL1]], [[SCALAR_PH]] ]
+; CHECK-EPI-TF-NEXT:    [[Q:%.*]] = phi ptr [ [[VAR4:%.*]], [[FOR_BODY]] ], [ [[BC_RESUME_VAL2]], [[SCALAR_PH]] ]
+; CHECK-EPI-TF-NEXT:    [[VAR0:%.*]] = phi i32 [ [[VAR2:%.*]], [[FOR_BODY]] ], [ [[BC_MERGE_RDX]], [[SCALAR_PH]] ]
+; CHECK-EPI-TF-NEXT:    [[VAR1:%.*]] = load i32, ptr [[P]], align 8
+; CHECK-EPI-TF-NEXT:    [[VAR2]] = add i32 [[VAR1]], [[VAR0]]
+; CHECK-EPI-TF-NEXT:    store ptr [[P]], ptr [[Q]], align 8
+; CHECK-EPI-TF-NEXT:    [[VAR3]] = getelementptr inbounds i32, ptr [[P]], i32 1
+; CHECK-EPI-TF-NEXT:    [[VAR4]] = getelementptr inbounds ptr, ptr [[Q]], i32 1
+; CHECK-EPI-TF-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 1
+; CHECK-EPI-TF-NEXT:    [[COND:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-EPI-TF-NEXT:    br i1 [[COND]], label [[FOR_BODY]], label [[FOR_END]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK-EPI-TF:       for.end:
+; CHECK-EPI-TF-NEXT:    [[VAR5:%.*]] = phi i32 [ [[VAR2]], [[FOR_BODY]] ], [ [[TMP14]], [[MIDDLE_BLOCK]] ]
+; CHECK-EPI-TF-NEXT:    ret i32 [[VAR5]]
+;
 entry:
   br label %for.body
 
@@ -313,6 +560,33 @@ define void @phi_used_in_vector_compare_and_scalar_indvar_update_and_store(ptr %
 ; CHECK:       for.end:
 ; CHECK-NEXT:    ret void
 ;
+; CHECK-EPI-TF-LABEL: @phi_used_in_vector_compare_and_scalar_indvar_update_and_store(
+; CHECK-EPI-TF-NEXT:  entry:
+; CHECK-EPI-TF-NEXT:    br label [[VECTOR_PH:%.*]]
+; CHECK-EPI-TF:       vector.ph:
+; CHECK-EPI-TF-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-EPI-TF-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP2:%.*]] = getelementptr i8, ptr [[PTR:%.*]], i64 2048
+; CHECK-EPI-TF-NEXT:    br label [[VECTOR_BODY:%.*]]
+; CHECK-EPI-TF:       vector.body:
+; CHECK-EPI-TF-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[POINTER_PHI:%.*]] = phi ptr [ [[PTR]], [[VECTOR_PH]] ], [ [[PTR_IND:%.*]], [[VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP3:%.*]] = call <vscale x 2 x i64> @llvm.stepvector.nxv2i64()
+; CHECK-EPI-TF-NEXT:    [[TMP4:%.*]] = shl <vscale x 2 x i64> [[TMP3]], splat (i64 1)
+; CHECK-EPI-TF-NEXT:    [[VECTOR_GEP:%.*]] = getelementptr i8, ptr [[POINTER_PHI]], <vscale x 2 x i64> [[TMP4]]
+; CHECK-EPI-TF-NEXT:    [[TMP5:%.*]] = extractelement <vscale x 2 x ptr> [[VECTOR_GEP]], i64 0
+; CHECK-EPI-TF-NEXT:    [[TMP6:%.*]] = icmp ne <vscale x 2 x ptr> [[VECTOR_GEP]], splat (ptr null)
+; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i16.p0(<vscale x 2 x i16> zeroinitializer, ptr align 2 [[TMP5]], <vscale x 2 x i1> [[TMP6]])
+; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; CHECK-EPI-TF-NEXT:    [[TMP7:%.*]] = shl i64 [[TMP1]], 1
+; CHECK-EPI-TF-NEXT:    [[PTR_IND]] = getelementptr i8, ptr [[POINTER_PHI]], i64 [[TMP7]]
+; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP8]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK-EPI-TF:       middle.block:
+; CHECK-EPI-TF-NEXT:    br label [[FOR_END:%.*]]
+; CHECK-EPI-TF:       for.end:
+; CHECK-EPI-TF-NEXT:    ret void
+;
 entry:
   br label %for.body
 

>From ecb81c70c0dcf626034b32f7aebbb858c73622de Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Sat, 19 Sep 2026 13:44:06 +0000
Subject: [PATCH 18/25] patch out support to reduction and fixed-order
 recurrence to minimize the patch

---
 .../Transforms/Vectorize/LoopVectorize.cpp    |  12 +-
 .../Transforms/Vectorize/VPlanLowering.cpp    |  27 +-
 .../AArch64/fold-epilogue-tail-reductions.ll  | 884 ------------------
 .../AArch64/fold-epilogue-tail.ll             | 277 ------
 .../AArch64/partial-reduce-with-predicate.ll  | 721 +-------------
 ...g-vectorization-fixed-order-recurrences.ll | 197 +---
 6 files changed, 25 insertions(+), 2093 deletions(-)
 delete mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index e0012aa9628a9..fa2b0f1e59038 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3401,7 +3401,7 @@ static bool hasFindLastReductionPhi(VPlan &Plan) {
 /// otherwise CM_EpilogueAllowed.
 static EpilogueLowering getEpilogueTailLowering(
     const LoopVectorizationCostModel &MainCM, const Loop *L,
-    OptimizationRemarkEmitter *ORE, const LoopVectorizationLegality &LVL,
+    OptimizationRemarkEmitter *ORE, LoopVectorizationLegality &LVL,
     const LoopVectorizeHints &Hints, const TargetTransformInfo *TTI) {
   // Epilogue TF is only enabled when explicitly requested via command line.
   if (!EpilogueTailFoldingPolicy.getNumOccurrences() ||
@@ -6531,7 +6531,8 @@ static bool verifyExecutionFrequenciesMatchBFI(VPlan &Plan, Loop *OrigLoop,
 }
 #endif
 
-VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(LoopVectorizationCostModel &EnabledCM) {
+VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(
+    LoopVectorizationCostModel &EnabledCM) {
   bool IsInnerLoop = OrigLoop->isInnermost();
 
   // Set up loop versioning for inner loops with memory runtime checks.
@@ -7689,11 +7690,6 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
           "active.lane.mask.entry");
       cast<VPHeaderPHIRecipe>(&R)->setStartValue(EntryALM);
       continue;
-    } else if (isa<VPFirstOrderRecurrencePHIRecipe>(&R)) {
-      auto *RecPhi = cast<VPFirstOrderRecurrencePHIRecipe>(&R);
-      VPInstruction *ResumeForEpi =
-          IRPhiToResumeForEpi.at(cast<PHINode>(RecPhi->getUnderlyingInstr()));
-      ResumeV = ResumeForEpi->getUnderlyingValue();
     } else {
       // Retrieve the induction resume value via ResumeForEpilogue.
       PHINode *IndPhi = cast<VPWidenInductionRecipe>(&R)->getPHINode();
@@ -8282,7 +8278,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
 
   // Destroy the cost model before executing any plan, so that code generation
   // cannot rely on cost-modeling decisions.
-  // LVP.clearCostModel();
+  LVP.clearCostModel();
 
   VPlan &BestPlan = *BestPlanPtr;
   // Consider vectorizing the epilogue too if it's profitable.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
index b724084bf2a97..89b6f79067180 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
@@ -47,8 +47,24 @@ void VPlanTransforms::replaceWideCanonicalIVWithWideIV(
 
   VPWidenCanonicalIVRecipe *WideCanIV = nullptr;
   VPIRValue *StartValue = nullptr;
-  // VPWidenCanonicalIVRecipe is either a direct user of CanonicalIV or
-  // Add (CanonicalIV, resumeValue) (like the case for tail-folded epilogue).
+  // Find VPWidenCanonicalIVRecipe among the canonical IV's users,
+  // matching Case 1 where it's a direct user of canonicalIV:
+  // <x1> vector loop: {
+  //   vp<%4> = CANONICAL-IV
+  //
+  //   vector.body:
+  //     EMIT vp<%6> = WIDEN-CANONICAL-INDUCTION nuw vp<%4>
+  //     ..
+  // }
+  // or Case 2 (indirect use through an epilogue resume-value add):
+  // <x1> vector loop: {
+  //   vp<%5> = CANONICAL-IV
+  //
+  //   vec.epilog.vector.body:
+  //     EMIT vp<%7> = add vp<%5>, ir<%vec.epilog.resume.val>
+  //     EMIT vp<%8> = WIDEN-CANONICAL-INDUCTION nuw vp<%7>
+  //     ..
+  //}
   auto *IV = LoopRegion->getCanonicalIV();
   for (auto *User : IV->users()) {
     if (isa<VPWidenCanonicalIVRecipe>(User)) {
@@ -580,12 +596,7 @@ void VPlanTransforms::convertToConcreteRecipes(VPlan &Plan) {
       }
 
       if (auto *WideCanIV = dyn_cast<VPWidenCanonicalIVRecipe>(&R)) {
-        VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
-        if (!LoopRegion)
-          continue;
-        VPValue *CanIV = LoopRegion->getCanonicalIV();
-        if (!CanIV)
-          continue;
+        VPValue *CanIV = WideCanIV->getCanonicalIV();
         Type *CanIVTy = CanIV->getScalarType();
         VPValue *Step = WideCanIV->getStepValue();
         if (!Step) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll
deleted file mode 100644
index 4323230e2b525..0000000000000
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail-reductions.ll
+++ /dev/null
@@ -1,884 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; REQUIRES: asserts
-; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 -mattr=+sve -S %s | FileCheck %s
-
-; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width="vscale x 16" -epilogue-vectorization-force-VF="vscale x 8" -mattr=+sve -S %s | FileCheck %s --check-prefix=CHECK-VS
-
-target triple = "aarch64-linux-gnu"
-
-define i32 @add_redc(ptr %src, i64 %n) {
-; CHECK-LABEL: define i32 @add_redc(
-; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 16
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 1
-; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i32>, ptr [[TMP3]], align 1
-; CHECK-NEXT:    [[TMP4]] = add <16 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
-; CHECK-NEXT:    [[TMP5]] = add <16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[BIN_RDX:%.*]] = add <16 x i32> [[TMP5]], [[TMP4]]
-; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.add.v16i32(<16 x i32> [[BIN_RDX]])
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
-; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i32> [ [[TMP8]], %[[VEC_EPILOG_PH]] ], [ [[TMP11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
-; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[TMP9]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
-; CHECK-NEXT:    [[TMP10:%.*]] = add <8 x i32> [[WIDE_MASKED_LOAD]], [[VEC_PHI5]]
-; CHECK-NEXT:    [[TMP11]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP10]], <8 x i32> [[VEC_PHI5]]
-; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-NEXT:    [[TMP13:%.*]] = xor i1 [[TMP12]], true
-; CHECK-NEXT:    br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP11]])
-; CHECK-NEXT:    br label %[[EXIT]]
-; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-NEXT:    br label %[[LOOP:.*]]
-; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[ADD:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[IV]]
-; CHECK-NEXT:    [[LOAD:%.*]] = load i32, ptr [[GEP]], align 1
-; CHECK-NEXT:    [[ADD]] = add i32 [[LOAD]], [[RED]]
-; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
-; CHECK-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
-; CHECK-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[ADD]], %[[LOOP]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[TMP14]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-NEXT:    ret i32 [[ADD_LCSSA]]
-;
-; CHECK-VS-LABEL: define i32 @add_redc(
-; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
-; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
-; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-VS:       [[VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP4]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 1
-; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
-; CHECK-VS-NEXT:    [[TMP8]] = add <vscale x 16 x i32> [[WIDE_LOAD]], [[VEC_PHI]]
-; CHECK-VS-NEXT:    [[TMP9]] = add <vscale x 16 x i32> [[WIDE_LOAD3]], [[VEC_PHI2]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-VS:       [[MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[BIN_RDX:%.*]] = add <vscale x 16 x i32> [[TMP9]], [[TMP8]]
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i32 @llvm.vector.reduce.add.nxv16i32(<vscale x 16 x i32> [[BIN_RDX]])
-; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = insertelement <vscale x 8 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i32> [ [[TMP14]], %[[VEC_EPILOG_PH]] ], [ [[TMP17:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
-; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 1 [[TMP15]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
-; CHECK-VS-NEXT:    [[TMP16:%.*]] = add <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], [[VEC_PHI5]]
-; CHECK-VS-NEXT:    [[TMP17]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> [[TMP16]], <vscale x 8 x i32> [[VEC_PHI5]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP13]]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
-; CHECK-VS-NEXT:    [[TMP18:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP19:%.*]] = xor i1 [[TMP18]], true
-; CHECK-VS-NEXT:    br i1 [[TMP19]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP20:%.*]] = call i32 @llvm.vector.reduce.add.nxv8i32(<vscale x 8 x i32> [[TMP17]])
-; CHECK-VS-NEXT:    br label %[[EXIT]]
-; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
-; CHECK-VS:       [[LOOP]]:
-; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-VS-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[ADD:%.*]], %[[LOOP]] ]
-; CHECK-VS-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[IV]]
-; CHECK-VS-NEXT:    [[LOAD:%.*]] = load i32, ptr [[GEP]], align 1
-; CHECK-VS-NEXT:    [[ADD]] = add i32 [[LOAD]], [[RED]]
-; CHECK-VS-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
-; CHECK-VS-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
-; CHECK-VS-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
-; CHECK-VS:       [[EXIT]]:
-; CHECK-VS-NEXT:    [[ADD_LCSSA:%.*]] = phi i32 [ [[ADD]], %[[LOOP]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ], [ [[TMP20]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-VS-NEXT:    ret i32 [[ADD_LCSSA]]
-;
-entry:
-  br label %loop
-
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %red = phi i32 [ 0, %entry ], [ %add, %loop ]
-  %gep = getelementptr inbounds i32, ptr %src, i64 %iv
-  %load = load i32, ptr %gep, align 1
-  %add = add i32 %load, %red
-  %iv.next = add i64 %iv, 1
-  %icmp3 = icmp eq i64 %iv, %n
-  br i1 %icmp3, label %exit, label %loop
-
-exit:
-  ret i32 %add
-}
-
-define i32 @max_redc(ptr %src, i64 %n) {
-; CHECK-LABEL: define i32 @max_redc(
-; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[TMP2]], i64 16
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 1
-; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i32>, ptr [[TMP3]], align 1
-; CHECK-NEXT:    [[TMP4]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD]], <16 x i32> [[VEC_PHI]])
-; CHECK-NEXT:    [[TMP5]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[WIDE_LOAD3]], <16 x i32> [[VEC_PHI2]])
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[RDX_MINMAX:%.*]] = call <16 x i32> @llvm.umax.v16i32(<16 x i32> [[TMP4]], <16 x i32> [[TMP5]])
-; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.umax.v16i32(<16 x i32> [[RDX_MINMAX]])
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i32> poison, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT]], <8 x i32> poison, <8 x i32> zeroinitializer
-; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i32> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
-; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 1 [[TMP8]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
-; CHECK-NEXT:    [[TMP9:%.*]] = call <8 x i32> @llvm.umax.v8i32(<8 x i32> [[WIDE_MASKED_LOAD]], <8 x i32> [[VEC_PHI5]])
-; CHECK-NEXT:    [[TMP10]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP9]], <8 x i32> [[VEC_PHI5]]
-; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-NEXT:    [[TMP12:%.*]] = xor i1 [[TMP11]], true
-; CHECK-NEXT:    br i1 [[TMP12]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[TMP13:%.*]] = call i32 @llvm.vector.reduce.umax.v8i32(<8 x i32> [[TMP10]])
-; CHECK-NEXT:    br label %[[EXIT]]
-; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-NEXT:    br label %[[LOOP:.*]]
-; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MAX:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[IV]]
-; CHECK-NEXT:    [[LOAD:%.*]] = load i32, ptr [[GEP]], align 1
-; CHECK-NEXT:    [[MAX]] = call i32 @llvm.umax.i32(i32 [[LOAD]], i32 [[RED]])
-; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
-; CHECK-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
-; CHECK-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[MAX]], %[[LOOP]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[TMP13]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-NEXT:    ret i32 [[MAX_LCSSA]]
-;
-; CHECK-VS-LABEL: define i32 @max_redc(
-; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
-; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
-; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-VS:       [[VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP9:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i32, ptr [[TMP6]], i64 [[TMP4]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP6]], align 1
-; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i32>, ptr [[TMP7]], align 1
-; CHECK-VS-NEXT:    [[TMP8]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD]], <vscale x 16 x i32> [[VEC_PHI]])
-; CHECK-VS-NEXT:    [[TMP9]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[WIDE_LOAD3]], <vscale x 16 x i32> [[VEC_PHI2]])
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-VS:       [[MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[RDX_MINMAX:%.*]] = call <vscale x 16 x i32> @llvm.umax.nxv16i32(<vscale x 16 x i32> [[TMP8]], <vscale x 16 x i32> [[TMP9]])
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i32 @llvm.vector.reduce.umax.nxv16i32(<vscale x 16 x i32> [[RDX_MINMAX]])
-; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 3
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i32> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP16:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[INDEX4]]
-; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 1 [[TMP14]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = call <vscale x 8 x i32> @llvm.umax.nxv8i32(<vscale x 8 x i32> [[WIDE_MASKED_LOAD]], <vscale x 8 x i32> [[VEC_PHI5]])
-; CHECK-VS-NEXT:    [[TMP16]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> [[TMP15]], <vscale x 8 x i32> [[VEC_PHI5]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP13]]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
-; CHECK-VS-NEXT:    [[TMP17:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
-; CHECK-VS-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP19:%.*]] = call i32 @llvm.vector.reduce.umax.nxv8i32(<vscale x 8 x i32> [[TMP16]])
-; CHECK-VS-NEXT:    br label %[[EXIT]]
-; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
-; CHECK-VS:       [[LOOP]]:
-; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-VS-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MAX:%.*]], %[[LOOP]] ]
-; CHECK-VS-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[SRC]], i64 [[IV]]
-; CHECK-VS-NEXT:    [[LOAD:%.*]] = load i32, ptr [[GEP]], align 1
-; CHECK-VS-NEXT:    [[MAX]] = call i32 @llvm.umax.i32(i32 [[LOAD]], i32 [[RED]])
-; CHECK-VS-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
-; CHECK-VS-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
-; CHECK-VS-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
-; CHECK-VS:       [[EXIT]]:
-; CHECK-VS-NEXT:    [[MAX_LCSSA:%.*]] = phi i32 [ [[MAX]], %[[LOOP]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ], [ [[TMP19]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-VS-NEXT:    ret i32 [[MAX_LCSSA]]
-;
-entry:
-  br label %loop
-
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %red = phi i32 [ 0, %entry ], [ %max, %loop ]
-  %gep = getelementptr inbounds i32, ptr %src, i64 %iv
-  %load = load i32, ptr %gep, align 1
-  %max = call i32 @llvm.umax(i32 %load, i32 %red)
-  %iv.next = add i64 %iv, 1
-  %icmp3 = icmp eq i64 %iv, %n
-  br i1 %icmp3, label %exit, label %loop
-
-exit:
-  ret i32 %max
-}
-
-define i64 @find_iv(ptr %src, i64 %n) {
-;
-; CHECK-LABEL: define i64 @find_iv(
-; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:  [[ENTRY:.*]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 16
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 15
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <16 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI1:%.*]] = phi <16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
-; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT:    [[TMP4]] = or <16 x i1> [[VEC_PHI1]], [[TMP3]]
-; CHECK-NEXT:    [[TMP5]] = select <16 x i1> [[TMP3]], <16 x i64> [[VEC_IND]], <16 x i64> [[VEC_PHI]]
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add <16 x i64> [[VEC_IND]], splat (i64 16)
-; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[TMP7:%.*]] = call i64 @llvm.vector.reduce.umax.v16i64(<16 x i64> [[TMP5]])
-; CHECK-NEXT:    [[TMP8:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[TMP4]])
-; CHECK-NEXT:    [[TMP9:%.*]] = freeze i1 [[TMP8]]
-; CHECK-NEXT:    [[RDX_SELECT:%.*]] = select i1 [[TMP9]], i64 [[TMP7]], i64 0
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
-; CHECK:       [[SCALAR_PH]]:
-; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT:    br label %[[LOOP:.*]]
-; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[RED:%.*]] = phi i64 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[SELECT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
-; CHECK-NEXT:    [[LOAD:%.*]] = load i8, ptr [[GEP]], align 1
-; CHECK-NEXT:    [[ICMP:%.*]] = icmp eq i8 [[LOAD]], 0
-; CHECK-NEXT:    [[SELECT]] = select i1 [[ICMP]], i64 [[IV]], i64 [[RED]]
-; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
-; CHECK-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
-; CHECK-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[SELECT_LCSSA:%.*]] = phi i64 [ [[SELECT]], %[[LOOP]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
-; CHECK-NEXT:    ret i64 [[SELECT_LCSSA]]
-;
-; CHECK-VS-LABEL: define i64 @find_iv(
-; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-VS-NEXT:  [[ENTRY:.*]]:
-; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 4
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
-; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP2]]
-; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = call <vscale x 16 x i64> @llvm.stepvector.nxv16i64()
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i64> poison, i64 [[TMP2]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i64> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-VS:       [[VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 16 x i64> [ [[TMP4]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i64> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP8:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI1:%.*]] = phi <vscale x 16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i8>, ptr [[TMP5]], align 1
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = icmp eq <vscale x 16 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-VS-NEXT:    [[TMP7]] = or <vscale x 16 x i1> [[VEC_PHI1]], [[TMP6]]
-; CHECK-VS-NEXT:    [[TMP8]] = select <vscale x 16 x i1> [[TMP6]], <vscale x 16 x i64> [[VEC_IND]], <vscale x 16 x i64> [[VEC_PHI]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; CHECK-VS-NEXT:    [[VEC_IND_NEXT]] = add <vscale x 16 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
-; CHECK-VS-NEXT:    [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-VS:       [[MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = call i64 @llvm.vector.reduce.umax.nxv16i64(<vscale x 16 x i64> [[TMP8]])
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i1 @llvm.vector.reduce.or.nxv16i1(<vscale x 16 x i1> [[TMP7]])
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = freeze i1 [[TMP11]]
-; CHECK-VS-NEXT:    [[RDX_SELECT:%.*]] = select i1 [[TMP12]], i64 [[TMP10]], i64 0
-; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
-; CHECK-VS:       [[SCALAR_PH]]:
-; CHECK-VS-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
-; CHECK-VS:       [[LOOP]]:
-; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-VS-NEXT:    [[RED:%.*]] = phi i64 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[SELECT:%.*]], %[[LOOP]] ]
-; CHECK-VS-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
-; CHECK-VS-NEXT:    [[LOAD:%.*]] = load i8, ptr [[GEP]], align 1
-; CHECK-VS-NEXT:    [[ICMP:%.*]] = icmp eq i8 [[LOAD]], 0
-; CHECK-VS-NEXT:    [[SELECT]] = select i1 [[ICMP]], i64 [[IV]], i64 [[RED]]
-; CHECK-VS-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
-; CHECK-VS-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
-; CHECK-VS-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
-; CHECK-VS:       [[EXIT]]:
-; CHECK-VS-NEXT:    [[SELECT_LCSSA:%.*]] = phi i64 [ [[SELECT]], %[[LOOP]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
-; CHECK-VS-NEXT:    ret i64 [[SELECT_LCSSA]]
-;
-entry:
-  br label %loop
-
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %red = phi i64 [ 0, %entry ], [ %select, %loop ]
-  %gep = getelementptr inbounds i8, ptr %src, i64 %iv
-  %load = load i8, ptr %gep, align 1
-  %icmp = icmp eq i8 %load, 0
-  %select = select i1 %icmp, i64 %iv, i64 %red
-  %iv.next = add i64 %iv, 1
-  %icmp3 = icmp eq i64 %iv, %n
-  br i1 %icmp3, label %exit, label %loop
-
-exit:
-  ret i64 %select
-}
-
-define i32 @any-of(ptr %src, i64 %n) {
-;
-; CHECK-LABEL: define i32 @any-of(
-; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], 32
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = and i64 [[TMP0]], 31
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP6:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i8, ptr [[TMP2]], i64 16
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
-; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP3]], align 1
-; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq <16 x i8> [[WIDE_LOAD3]], zeroinitializer
-; CHECK-NEXT:    [[TMP6]] = or <16 x i1> [[VEC_PHI]], [[TMP4]]
-; CHECK-NEXT:    [[TMP7]] = or <16 x i1> [[VEC_PHI2]], [[TMP5]]
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[BIN_RDX:%.*]] = or <16 x i1> [[TMP7]], [[TMP6]]
-; CHECK-NEXT:    [[TMP9:%.*]] = call i1 @llvm.vector.reduce.or.v16i1(<16 x i1> [[BIN_RDX]])
-; CHECK-NEXT:    [[TMP10:%.*]] = freeze i1 [[TMP9]]
-; CHECK-NEXT:    [[RDX_SELECT:%.*]] = select i1 [[TMP10]], i32 1, i32 0
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[TMP11:%.*]] = icmp ne i32 [[BC_MERGE_RDX]], 0
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <8 x i1> poison, i1 [[TMP11]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <8 x i1> [[BROADCAST_SPLATINSERT]], <8 x i1> poison, <8 x i32> zeroinitializer
-; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI5:%.*]] = phi <8 x i1> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP15:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP12:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX4]]
-; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP12]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
-; CHECK-NEXT:    [[TMP13:%.*]] = icmp eq <8 x i8> [[WIDE_MASKED_LOAD]], zeroinitializer
-; CHECK-NEXT:    [[TMP14:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP13]], <8 x i1> zeroinitializer
-; CHECK-NEXT:    [[TMP15]] = or <8 x i1> [[VEC_PHI5]], [[TMP14]]
-; CHECK-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
-; CHECK-NEXT:    [[TMP16:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-NEXT:    [[TMP17:%.*]] = xor i1 [[TMP16]], true
-; CHECK-NEXT:    br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[TMP18:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP15]])
-; CHECK-NEXT:    [[TMP19:%.*]] = freeze i1 [[TMP18]]
-; CHECK-NEXT:    [[RDX_SELECT7:%.*]] = select i1 [[TMP19]], i32 1, i32 0
-; CHECK-NEXT:    br label %[[EXIT]]
-; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-NEXT:    br label %[[LOOP:.*]]
-; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SELECT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
-; CHECK-NEXT:    [[LOAD:%.*]] = load i8, ptr [[GEP]], align 1
-; CHECK-NEXT:    [[ICMP:%.*]] = icmp eq i8 [[LOAD]], 0
-; CHECK-NEXT:    [[SELECT]] = select i1 [[ICMP]], i32 1, i32 [[RED]]
-; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
-; CHECK-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
-; CHECK-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[SELECT_LCSSA:%.*]] = phi i32 [ [[SELECT]], %[[LOOP]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ [[RDX_SELECT7]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-NEXT:    ret i32 [[SELECT_LCSSA]]
-;
-; CHECK-VS-LABEL: define i32 @any-of(
-; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i64 [[N]], 1
-; CHECK-VS-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP1]], 5
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[TMP0]], [[TMP3]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP1]], 4
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[TMP0]], [[TMP3]]
-; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[TMP0]], [[N_MOD_VF]]
-; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-VS:       [[VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP10:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI2:%.*]] = phi <vscale x 16 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP11:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i8, ptr [[TMP6]], i64 [[TMP4]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i8>, ptr [[TMP6]], align 1
-; CHECK-VS-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 16 x i8>, ptr [[TMP7]], align 1
-; CHECK-VS-NEXT:    [[TMP8:%.*]] = icmp eq <vscale x 16 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-VS-NEXT:    [[TMP9:%.*]] = icmp eq <vscale x 16 x i8> [[WIDE_LOAD3]], zeroinitializer
-; CHECK-VS-NEXT:    [[TMP10]] = or <vscale x 16 x i1> [[VEC_PHI]], [[TMP8]]
-; CHECK-VS-NEXT:    [[TMP11]] = or <vscale x 16 x i1> [[VEC_PHI2]], [[TMP9]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP3]]
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-VS:       [[MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[BIN_RDX:%.*]] = or <vscale x 16 x i1> [[TMP11]], [[TMP10]]
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = call i1 @llvm.vector.reduce.or.nxv16i1(<vscale x 16 x i1> [[BIN_RDX]])
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = freeze i1 [[TMP13]]
-; CHECK-VS-NEXT:    [[RDX_SELECT:%.*]] = select i1 [[TMP14]], i32 1, i32 0
-; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = icmp ne i32 [[BC_MERGE_RDX]], 0
-; CHECK-VS-NEXT:    [[TMP16:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP17:%.*]] = shl nuw i64 [[TMP16]], 3
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[TMP0]])
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 8 x i1> poison, i1 [[TMP15]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 8 x i1> [[BROADCAST_SPLATINSERT]], <vscale x 8 x i1> poison, <vscale x 8 x i32> zeroinitializer
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT6:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI5:%.*]] = phi <vscale x 8 x i1> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[TMP21:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[INDEX4]]
-; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i8> @llvm.masked.load.nxv8i8.p0(ptr align 1 [[TMP18]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i8> poison)
-; CHECK-VS-NEXT:    [[TMP19:%.*]] = icmp eq <vscale x 8 x i8> [[WIDE_MASKED_LOAD]], zeroinitializer
-; CHECK-VS-NEXT:    [[TMP20:%.*]] = select <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i1> [[TMP19]], <vscale x 8 x i1> zeroinitializer
-; CHECK-VS-NEXT:    [[TMP21]] = or <vscale x 8 x i1> [[VEC_PHI5]], [[TMP20]]
-; CHECK-VS-NEXT:    [[INDEX_NEXT6]] = add i64 [[INDEX4]], [[TMP17]]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT6]], i64 [[TMP0]])
-; CHECK-VS-NEXT:    [[TMP22:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP23:%.*]] = xor i1 [[TMP22]], true
-; CHECK-VS-NEXT:    br i1 [[TMP23]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP24:%.*]] = call i1 @llvm.vector.reduce.or.nxv8i1(<vscale x 8 x i1> [[TMP21]])
-; CHECK-VS-NEXT:    [[TMP25:%.*]] = freeze i1 [[TMP24]]
-; CHECK-VS-NEXT:    [[RDX_SELECT7:%.*]] = select i1 [[TMP25]], i32 1, i32 0
-; CHECK-VS-NEXT:    br label %[[EXIT]]
-; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
-; CHECK-VS:       [[LOOP]]:
-; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-VS-NEXT:    [[RED:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SELECT:%.*]], %[[LOOP]] ]
-; CHECK-VS-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
-; CHECK-VS-NEXT:    [[LOAD:%.*]] = load i8, ptr [[GEP]], align 1
-; CHECK-VS-NEXT:    [[ICMP:%.*]] = icmp eq i8 [[LOAD]], 0
-; CHECK-VS-NEXT:    [[SELECT]] = select i1 [[ICMP]], i32 1, i32 [[RED]]
-; CHECK-VS-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
-; CHECK-VS-NEXT:    [[ICMP3:%.*]] = icmp eq i64 [[IV]], [[N]]
-; CHECK-VS-NEXT:    br i1 [[ICMP3]], label %[[EXIT]], label %[[LOOP]]
-; CHECK-VS:       [[EXIT]]:
-; CHECK-VS-NEXT:    [[SELECT_LCSSA:%.*]] = phi i32 [ [[SELECT]], %[[LOOP]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ [[RDX_SELECT7]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-VS-NEXT:    ret i32 [[SELECT_LCSSA]]
-;
-entry:
-  br label %loop
-
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %red = phi i32 [ 0, %entry ], [ %select, %loop ]
-  %gep = getelementptr inbounds i8, ptr %src, i64 %iv
-  %load = load i8, ptr %gep, align 1
-  %icmp = icmp eq i8 %load, 0
-  %select = select i1 %icmp, i32 1, i32 %red
-  %iv.next = add i64 %iv, 1
-  %icmp3 = icmp eq i64 %iv, %n
-  br i1 %icmp3, label %exit, label %loop
-
-exit:
-  ret i32 %select
-}
-
-define i64 @arg_min_first_index(ptr %arr, i64 %n, i64 %start) {
-; CHECK-LABEL: define i64 @arg_min_first_index(
-; CHECK-SAME: ptr [[ARR:%.*]], i64 [[N:%.*]], i64 [[START:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 16
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 15
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <16 x i64> poison, i64 [[START]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <16 x i64> [[BROADCAST_SPLATINSERT]], <16 x i64> poison, <16 x i32> zeroinitializer
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <16 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7, i64 8, i64 9, i64 10, i64 11, i64 12, i64 13, i64 14, i64 15>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <16 x i64> [ [[BROADCAST_SPLAT]], %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <16 x i64> [ poison, %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[INDEX]]
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i64>, ptr [[TMP1]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = icmp slt <16 x i64> [[WIDE_LOAD]], [[VEC_PHI]]
-; CHECK-NEXT:    [[TMP3]] = select <16 x i1> [[TMP2]], <16 x i64> [[VEC_IND]], <16 x i64> [[VEC_PHI2]]
-; CHECK-NEXT:    [[TMP4]] = call <16 x i64> @llvm.smin.v16i64(<16 x i64> [[WIDE_LOAD]], <16 x i64> [[VEC_PHI]])
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <16 x i64> [[VEC_IND]], splat (i64 16)
-; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP5]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[TMP6:%.*]] = call i64 @llvm.vector.reduce.smin.v16i64(<16 x i64> [[TMP4]])
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <16 x i64> poison, i64 [[TMP6]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <16 x i64> [[BROADCAST_SPLATINSERT3]], <16 x i64> poison, <16 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq <16 x i64> [[TMP4]], [[BROADCAST_SPLAT4]]
-; CHECK-NEXT:    [[TMP8:%.*]] = select <16 x i1> [[TMP7]], <16 x i64> [[TMP3]], <16 x i64> splat (i64 -1)
-; CHECK-NEXT:    [[TMP9:%.*]] = call i64 @llvm.vector.reduce.umin.v16i64(<16 x i64> [[TMP8]])
-; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[TMP6]], [[START]]
-; CHECK-NEXT:    [[TMP11:%.*]] = select i1 [[TMP10]], i64 0, i64 [[TMP9]]
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 8
-; CHECK-NEXT:    br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
-; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[TMP6]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[START]], %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[BC_MERGE_RDX5:%.*]] = phi i64 [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[TMP12:%.*]] = and i64 [[N]], 7
-; CHECK-NEXT:    [[N_VEC6:%.*]] = sub i64 [[N]], [[TMP12]]
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT7:%.*]] = insertelement <8 x i64> poison, i64 [[BC_MERGE_RDX]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT8:%.*]] = shufflevector <8 x i64> [[BROADCAST_SPLATINSERT7]], <8 x i64> poison, <8 x i32> zeroinitializer
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT9:%.*]] = insertelement <8 x i64> poison, i64 [[BC_MERGE_RDX5]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT10:%.*]] = shufflevector <8 x i64> [[BROADCAST_SPLATINSERT9]], <8 x i64> poison, <8 x i32> zeroinitializer
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT11:%.*]] = insertelement <8 x i64> poison, i64 [[VEC_EPILOG_RESUME_VAL]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT12:%.*]] = shufflevector <8 x i64> [[BROADCAST_SPLATINSERT11]], <8 x i64> poison, <8 x i32> zeroinitializer
-; CHECK-NEXT:    [[INDUCTION:%.*]] = add nuw nsw <8 x i64> [[BROADCAST_SPLAT12]], <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7>
-; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX13:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT18:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_IND14:%.*]] = phi <8 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT19:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI15:%.*]] = phi <8 x i64> [ [[BROADCAST_SPLAT8]], %[[VEC_EPILOG_PH]] ], [ [[TMP16:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI16:%.*]] = phi <8 x i64> [ [[BROADCAST_SPLAT10]], %[[VEC_EPILOG_PH]] ], [ [[TMP15:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP13:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[INDEX13]]
-; CHECK-NEXT:    [[WIDE_LOAD17:%.*]] = load <8 x i64>, ptr [[TMP13]], align 8
-; CHECK-NEXT:    [[TMP14:%.*]] = icmp slt <8 x i64> [[WIDE_LOAD17]], [[VEC_PHI15]]
-; CHECK-NEXT:    [[TMP15]] = select <8 x i1> [[TMP14]], <8 x i64> [[VEC_IND14]], <8 x i64> [[VEC_PHI16]]
-; CHECK-NEXT:    [[TMP16]] = call <8 x i64> @llvm.smin.v8i64(<8 x i64> [[WIDE_LOAD17]], <8 x i64> [[VEC_PHI15]])
-; CHECK-NEXT:    [[INDEX_NEXT18]] = add nuw i64 [[INDEX13]], 8
-; CHECK-NEXT:    [[VEC_IND_NEXT19]] = add nuw nsw <8 x i64> [[VEC_IND14]], splat (i64 8)
-; CHECK-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT18]], [[N_VEC6]]
-; CHECK-NEXT:    br i1 [[TMP17]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[TMP18:%.*]] = call i64 @llvm.vector.reduce.smin.v8i64(<8 x i64> [[TMP16]])
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT20:%.*]] = insertelement <8 x i64> poison, i64 [[TMP18]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT21:%.*]] = shufflevector <8 x i64> [[BROADCAST_SPLATINSERT20]], <8 x i64> poison, <8 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP19:%.*]] = icmp eq <8 x i64> [[TMP16]], [[BROADCAST_SPLAT21]]
-; CHECK-NEXT:    [[TMP20:%.*]] = select <8 x i1> [[TMP19]], <8 x i64> [[TMP15]], <8 x i64> splat (i64 -1)
-; CHECK-NEXT:    [[TMP21:%.*]] = call i64 @llvm.vector.reduce.umin.v8i64(<8 x i64> [[TMP20]])
-; CHECK-NEXT:    [[TMP22:%.*]] = icmp eq i64 [[TMP18]], [[START]]
-; CHECK-NEXT:    [[TMP23:%.*]] = select i1 [[TMP22]], i64 0, i64 [[TMP21]]
-; CHECK-NEXT:    [[CMP_N22:%.*]] = icmp eq i64 [[N]], [[N_VEC6]]
-; CHECK-NEXT:    br i1 [[CMP_N22]], label %[[FOR_END]], label %[[VEC_EPILOG_SCALAR_PH]]
-; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC6]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
-; CHECK-NEXT:    [[BC_MERGE_RDX23:%.*]] = phi i64 [ [[TMP18]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP6]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[START]], %[[ITER_CHECK]] ]
-; CHECK-NEXT:    [[BC_MERGE_RDX24:%.*]] = phi i64 [ [[TMP23]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP11]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
-; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
-; CHECK:       [[FOR_BODY]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; CHECK-NEXT:    [[MIN:%.*]] = phi i64 [ [[BC_MERGE_RDX23]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MIN_NEXT:%.*]], %[[FOR_BODY]] ]
-; CHECK-NEXT:    [[MIN_LOC:%.*]] = phi i64 [ [[BC_MERGE_RDX24]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MIN_LOC_NEXT:%.*]], %[[FOR_BODY]] ]
-; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[IV]]
-; CHECK-NEXT:    [[TMP24:%.*]] = load i64, ptr [[ARRAYIDX]], align 8
-; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i64 [[TMP24]], [[MIN]]
-; CHECK-NEXT:    [[MIN_LOC_NEXT]] = select i1 [[CMP]], i64 [[IV]], i64 [[MIN_LOC]]
-; CHECK-NEXT:    [[MIN_NEXT]] = tail call i64 @llvm.smin.i64(i64 [[TMP24]], i64 [[MIN]])
-; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EXITCOND_NOT]], label %[[FOR_END]], label %[[FOR_BODY]]
-; CHECK:       [[FOR_END]]:
-; CHECK-NEXT:    [[MIN_LOC_NEXT_LCSSA:%.*]] = phi i64 [ [[MIN_LOC_NEXT]], %[[FOR_BODY]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ], [ [[TMP23]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-NEXT:    ret i64 [[MIN_LOC_NEXT_LCSSA]]
-;
-; CHECK-VS-LABEL: define i64 @arg_min_first_index(
-; CHECK-VS-SAME: ptr [[ARR:%.*]], i64 [[N:%.*]], i64 [[START:%.*]]) #[[ATTR0]] {
-; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 4
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
-; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
-; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i64> poison, i64 [[START]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i64> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = call <vscale x 16 x i64> @llvm.stepvector.nxv16i64()
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <vscale x 16 x i64> poison, i64 [[TMP2]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <vscale x 16 x i64> [[BROADCAST_SPLATINSERT2]], <vscale x 16 x i64> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-VS:       [[VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 16 x i64> [ [[TMP3]], %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI:%.*]] = phi <vscale x 16 x i64> [ [[BROADCAST_SPLAT]], %[[VECTOR_PH]] ], [ [[TMP7:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI4:%.*]] = phi <vscale x 16 x i64> [ poison, %[[VECTOR_PH]] ], [ [[TMP6:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i64>, ptr [[TMP4]], align 8
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = icmp slt <vscale x 16 x i64> [[WIDE_LOAD]], [[VEC_PHI]]
-; CHECK-VS-NEXT:    [[TMP6]] = select <vscale x 16 x i1> [[TMP5]], <vscale x 16 x i64> [[VEC_IND]], <vscale x 16 x i64> [[VEC_PHI4]]
-; CHECK-VS-NEXT:    [[TMP7]] = call <vscale x 16 x i64> @llvm.smin.nxv16i64(<vscale x 16 x i64> [[WIDE_LOAD]], <vscale x 16 x i64> [[VEC_PHI]])
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; CHECK-VS-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <vscale x 16 x i64> [[VEC_IND]], [[BROADCAST_SPLAT3]]
-; CHECK-VS-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-VS:       [[MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP9:%.*]] = call i64 @llvm.vector.reduce.smin.nxv16i64(<vscale x 16 x i64> [[TMP7]])
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT5:%.*]] = insertelement <vscale x 16 x i64> poison, i64 [[TMP9]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT6:%.*]] = shufflevector <vscale x 16 x i64> [[BROADCAST_SPLATINSERT5]], <vscale x 16 x i64> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = icmp eq <vscale x 16 x i64> [[TMP7]], [[BROADCAST_SPLAT6]]
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = select <vscale x 16 x i1> [[TMP10]], <vscale x 16 x i64> [[TMP6]], <vscale x 16 x i64> splat (i64 -1)
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vector.reduce.umin.nxv16i64(<vscale x 16 x i64> [[TMP11]])
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[TMP9]], [[START]]
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = select i1 [[TMP13]], i64 0, i64 [[TMP12]]
-; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], [[TMP1]]
-; CHECK-VS-NEXT:    br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
-; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[TMP9]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[START]], %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[BC_MERGE_RDX7:%.*]] = phi i64 [ [[TMP14]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP16:%.*]] = shl nuw i64 [[TMP15]], 3
-; CHECK-VS-NEXT:    [[N_MOD_VF8:%.*]] = urem i64 [[N]], [[TMP16]]
-; CHECK-VS-NEXT:    [[N_VEC9:%.*]] = sub i64 [[N]], [[N_MOD_VF8]]
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT10:%.*]] = insertelement <vscale x 8 x i64> poison, i64 [[BC_MERGE_RDX]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT11:%.*]] = shufflevector <vscale x 8 x i64> [[BROADCAST_SPLATINSERT10]], <vscale x 8 x i64> poison, <vscale x 8 x i32> zeroinitializer
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT12:%.*]] = insertelement <vscale x 8 x i64> poison, i64 [[BC_MERGE_RDX7]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT13:%.*]] = shufflevector <vscale x 8 x i64> [[BROADCAST_SPLATINSERT12]], <vscale x 8 x i64> poison, <vscale x 8 x i32> zeroinitializer
-; CHECK-VS-NEXT:    [[TMP17:%.*]] = call <vscale x 8 x i64> @llvm.stepvector.nxv8i64()
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT14:%.*]] = insertelement <vscale x 8 x i64> poison, i64 [[VEC_EPILOG_RESUME_VAL]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT15:%.*]] = shufflevector <vscale x 8 x i64> [[BROADCAST_SPLATINSERT14]], <vscale x 8 x i64> poison, <vscale x 8 x i32> zeroinitializer
-; CHECK-VS-NEXT:    [[INDUCTION:%.*]] = add nuw nsw <vscale x 8 x i64> [[BROADCAST_SPLAT15]], [[TMP17]]
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT16:%.*]] = insertelement <vscale x 8 x i64> poison, i64 [[TMP16]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT17:%.*]] = shufflevector <vscale x 8 x i64> [[BROADCAST_SPLATINSERT16]], <vscale x 8 x i64> poison, <vscale x 8 x i32> zeroinitializer
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX18:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT23:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_IND19:%.*]] = phi <vscale x 8 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT24:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI20:%.*]] = phi <vscale x 8 x i64> [ [[BROADCAST_SPLAT11]], %[[VEC_EPILOG_PH]] ], [ [[TMP21:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VEC_PHI21:%.*]] = phi <vscale x 8 x i64> [ [[BROADCAST_SPLAT13]], %[[VEC_EPILOG_PH]] ], [ [[TMP20:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP18:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[INDEX18]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD22:%.*]] = load <vscale x 8 x i64>, ptr [[TMP18]], align 8
-; CHECK-VS-NEXT:    [[TMP19:%.*]] = icmp slt <vscale x 8 x i64> [[WIDE_LOAD22]], [[VEC_PHI20]]
-; CHECK-VS-NEXT:    [[TMP20]] = select <vscale x 8 x i1> [[TMP19]], <vscale x 8 x i64> [[VEC_IND19]], <vscale x 8 x i64> [[VEC_PHI21]]
-; CHECK-VS-NEXT:    [[TMP21]] = call <vscale x 8 x i64> @llvm.smin.nxv8i64(<vscale x 8 x i64> [[WIDE_LOAD22]], <vscale x 8 x i64> [[VEC_PHI20]])
-; CHECK-VS-NEXT:    [[INDEX_NEXT23]] = add nuw i64 [[INDEX18]], [[TMP16]]
-; CHECK-VS-NEXT:    [[VEC_IND_NEXT24]] = add nuw nsw <vscale x 8 x i64> [[VEC_IND19]], [[BROADCAST_SPLAT17]]
-; CHECK-VS-NEXT:    [[TMP22:%.*]] = icmp eq i64 [[INDEX_NEXT23]], [[N_VEC9]]
-; CHECK-VS-NEXT:    br i1 [[TMP22]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP23:%.*]] = call i64 @llvm.vector.reduce.smin.nxv8i64(<vscale x 8 x i64> [[TMP21]])
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT25:%.*]] = insertelement <vscale x 8 x i64> poison, i64 [[TMP23]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT26:%.*]] = shufflevector <vscale x 8 x i64> [[BROADCAST_SPLATINSERT25]], <vscale x 8 x i64> poison, <vscale x 8 x i32> zeroinitializer
-; CHECK-VS-NEXT:    [[TMP24:%.*]] = icmp eq <vscale x 8 x i64> [[TMP21]], [[BROADCAST_SPLAT26]]
-; CHECK-VS-NEXT:    [[TMP25:%.*]] = select <vscale x 8 x i1> [[TMP24]], <vscale x 8 x i64> [[TMP20]], <vscale x 8 x i64> splat (i64 -1)
-; CHECK-VS-NEXT:    [[TMP26:%.*]] = call i64 @llvm.vector.reduce.umin.nxv8i64(<vscale x 8 x i64> [[TMP25]])
-; CHECK-VS-NEXT:    [[TMP27:%.*]] = icmp eq i64 [[TMP23]], [[START]]
-; CHECK-VS-NEXT:    [[TMP28:%.*]] = select i1 [[TMP27]], i64 0, i64 [[TMP26]]
-; CHECK-VS-NEXT:    [[CMP_N27:%.*]] = icmp eq i64 [[N]], [[N_VEC9]]
-; CHECK-VS-NEXT:    br i1 [[CMP_N27]], label %[[FOR_END]], label %[[VEC_EPILOG_SCALAR_PH]]
-; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-VS-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[BC_MERGE_RDX28:%.*]] = phi i64 [ [[TMP23]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP9]], %[[VEC_EPILOG_ITER_CHECK]] ], [ [[START]], %[[ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[BC_MERGE_RDX29:%.*]] = phi i64 [ [[TMP28]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[TMP14]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
-; CHECK-VS-NEXT:    br label %[[FOR_BODY:.*]]
-; CHECK-VS:       [[FOR_BODY]]:
-; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; CHECK-VS-NEXT:    [[MIN:%.*]] = phi i64 [ [[BC_MERGE_RDX28]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MIN_NEXT:%.*]], %[[FOR_BODY]] ]
-; CHECK-VS-NEXT:    [[MIN_LOC:%.*]] = phi i64 [ [[BC_MERGE_RDX29]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[MIN_LOC_NEXT:%.*]], %[[FOR_BODY]] ]
-; CHECK-VS-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw [8 x i8], ptr [[ARR]], i64 [[IV]]
-; CHECK-VS-NEXT:    [[TMP29:%.*]] = load i64, ptr [[ARRAYIDX]], align 8
-; CHECK-VS-NEXT:    [[CMP:%.*]] = icmp slt i64 [[TMP29]], [[MIN]]
-; CHECK-VS-NEXT:    [[MIN_LOC_NEXT]] = select i1 [[CMP]], i64 [[IV]], i64 [[MIN_LOC]]
-; CHECK-VS-NEXT:    [[MIN_NEXT]] = tail call i64 @llvm.smin.i64(i64 [[TMP29]], i64 [[MIN]])
-; CHECK-VS-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-VS-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-VS-NEXT:    br i1 [[EXITCOND_NOT]], label %[[FOR_END]], label %[[FOR_BODY]]
-; CHECK-VS:       [[FOR_END]]:
-; CHECK-VS-NEXT:    [[MIN_LOC_NEXT_LCSSA:%.*]] = phi i64 [ [[MIN_LOC_NEXT]], %[[FOR_BODY]] ], [ [[TMP14]], %[[MIDDLE_BLOCK]] ], [ [[TMP28]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-VS-NEXT:    ret i64 [[MIN_LOC_NEXT_LCSSA]]
-;
-entry:
-  br label %for.body
-
-for.body:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
-  %min = phi i64 [ %start, %entry ], [ %min.next, %for.body ]
-  %min.loc = phi i64 [ 0, %entry ], [ %min.loc.next, %for.body ]
-  %arrayidx = getelementptr inbounds nuw [8 x i8], ptr %arr, i64 %iv
-  %0 = load i64, ptr %arrayidx
-  %cmp = icmp slt i64 %0, %min
-  %min.loc.next = select i1 %cmp, i64 %iv, i64 %min.loc
-  %min.next = tail call i64 @llvm.smin.i64(i64 %0, i64 %min)
-  %iv.next = add nuw nsw i64 %iv, 1
-  %exitcond.not = icmp eq i64 %iv.next, %n
-  br i1 %exitcond.not, label %for.end, label %for.body
-
-for.end:
-  ret i64 %min.loc.next
-}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index a610939954df2..173297610c9dd 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -17,11 +17,6 @@
 ; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s \
 ; RUN: --check-prefix=CHECK-INVALID-COSTS
 
-; RUN: opt -p loop-vectorize -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-vector-width=8 -epilogue-vectorization-force-VF=4 \
-; RUN: -force-target-supports-masked-memory-ops -S %s | FileCheck %s \
-; RUN: --check-prefix=CHECK-INDUCTION
-
 ; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize,vectorutils \
 ; RUN: -epilogue-tail-folding-policy=prefer-fold-tail --disable-output \
 ; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 \
@@ -310,167 +305,6 @@ for.end:
   ret i32 %load
 }
 
-define i32 @live_out_recurrence(ptr %A, i64 %n) {
-; CHECK-LABEL: define i32 @live_out_recurrence(
-; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK:       [[VECTOR_PH]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
-; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 16
-; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK:       [[MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[VECTOR_RECUR_EXTRACT_FOR_PHI:%.*]] = extractelement <16 x i32> [[WIDE_LOAD]], i64 14
-; CHECK-NEXT:    [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <16 x i32> [[WIDE_LOAD]], i64 15
-; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[SCALAR_RECUR_INIT:%.*]] = phi i32 [ [[VECTOR_RECUR_EXTRACT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
-; CHECK-NEXT:    [[VECTOR_RECUR_INIT:%.*]] = insertelement <8 x i32> poison, i32 [[SCALAR_RECUR_INIT]], i32 7
-; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VECTOR_RECUR:%.*]] = phi <8 x i32> [ [[VECTOR_RECUR_INIT]], %[[VEC_EPILOG_PH]] ], [ [[WIDE_MASKED_LOAD:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX2]]
-; CHECK-NEXT:    [[WIDE_MASKED_LOAD]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
-; CHECK-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-NEXT:    [[TMP6:%.*]] = xor i1 [[TMP5]], true
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <8 x i32> [[VECTOR_RECUR]], <8 x i32> [[WIDE_MASKED_LOAD]], <8 x i32> <i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14>
-; CHECK-NEXT:    [[TMP8:%.*]] = xor <8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
-; CHECK-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v8i1(<8 x i1> [[TMP8]], i1 false)
-; CHECK-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
-; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <8 x i32> [[TMP7]], i64 [[LAST_ACTIVE_LANE]]
-; CHECK-NEXT:    br label %[[EXIT]]
-; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-NEXT:    br label %[[LOOP:.*]]
-; CHECK:       [[LOOP]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[FOR:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[L:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT:    [[L]] = load i32, ptr [[GEP]], align 4
-; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[FOR_LCSSA:%.*]] = phi i32 [ [[FOR]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ], [ [[TMP9]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-NEXT:    ret i32 [[FOR_LCSSA]]
-;
-; CHECK-VS-LABEL: define i32 @live_out_recurrence(
-; CHECK-VS-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 5
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
-; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
-; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-VS:       [[VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 [[TMP3]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP5]], align 4
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-VS:       [[MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vscale.i32()
-; CHECK-VS-NEXT:    [[TMP8:%.*]] = mul nuw i32 [[TMP7]], 16
-; CHECK-VS-NEXT:    [[TMP9:%.*]] = sub i32 [[TMP8]], 2
-; CHECK-VS-NEXT:    [[VECTOR_RECUR_EXTRACT_FOR_PHI:%.*]] = extractelement <vscale x 16 x i32> [[WIDE_LOAD]], i32 [[TMP9]]
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = call i32 @llvm.vscale.i32()
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = mul nuw i32 [[TMP10]], 16
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = sub i32 [[TMP11]], 1
-; CHECK-VS-NEXT:    [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <vscale x 16 x i32> [[WIDE_LOAD]], i32 [[TMP12]]
-; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[SCALAR_RECUR_INIT:%.*]] = phi i32 [ [[VECTOR_RECUR_EXTRACT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = shl nuw i64 [[TMP13]], 3
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = call i32 @llvm.vscale.i32()
-; CHECK-VS-NEXT:    [[TMP16:%.*]] = mul nuw i32 [[TMP15]], 8
-; CHECK-VS-NEXT:    [[TMP17:%.*]] = sub i32 [[TMP16]], 1
-; CHECK-VS-NEXT:    [[VECTOR_RECUR_INIT:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[SCALAR_RECUR_INIT]], i32 [[TMP17]]
-; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[VECTOR_RECUR:%.*]] = phi <vscale x 8 x i32> [ [[VECTOR_RECUR_INIT]], %[[VEC_EPILOG_PH]] ], [ [[WIDE_MASKED_LOAD:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX2]]
-; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 4 [[TMP18]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
-; CHECK-VS-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], [[TMP14]]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
-; CHECK-VS-NEXT:    [[TMP19:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP20:%.*]] = xor i1 [[TMP19]], true
-; CHECK-VS-NEXT:    br i1 [[TMP20]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP21:%.*]] = call <vscale x 8 x i32> @llvm.vector.splice.right.nxv8i32(<vscale x 8 x i32> [[VECTOR_RECUR]], <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], i32 1)
-; CHECK-VS-NEXT:    [[TMP22:%.*]] = xor <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
-; CHECK-VS-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP22]], i1 false)
-; CHECK-VS-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
-; CHECK-VS-NEXT:    [[TMP23:%.*]] = extractelement <vscale x 8 x i32> [[TMP21]], i64 [[LAST_ACTIVE_LANE]]
-; CHECK-VS-NEXT:    br label %[[EXIT]]
-; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
-; CHECK-VS:       [[LOOP]]:
-; CHECK-VS-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-VS-NEXT:    [[FOR:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[L:%.*]], %[[LOOP]] ]
-; CHECK-VS-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
-; CHECK-VS-NEXT:    [[L]] = load i32, ptr [[GEP]], align 4
-; CHECK-VS-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-VS-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-VS-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
-; CHECK-VS:       [[EXIT]]:
-; CHECK-VS-NEXT:    [[FOR_LCSSA:%.*]] = phi i32 [ [[FOR]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ], [ [[TMP23]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-VS-NEXT:    ret i32 [[FOR_LCSSA]]
-;
-entry:
-  br label %loop
-
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %for = phi i32 [ 0, %entry ], [ %l, %loop ]
-  %gep = getelementptr inbounds i32, ptr %A, i64 %iv
-  %l = load i32, ptr %gep, align 4
-  %iv.next = add nuw nsw i64 %iv, 1
-  %ec = icmp eq i64 %iv.next, %n
-  br i1 %ec, label %exit, label %loop
-
-exit:
-  ret i32 %for
-}
 
 define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-LABEL: define void @reversed-loop(
@@ -680,117 +514,6 @@ for.end:
 }
 declare void @foo(ptr)
 
-define i64 @find_last_offset_wide_canonical_iv(ptr %A, i64 %n) {
-; CHECK-INDUCTION-LABEL: define i64 @find_last_offset_wide_canonical_iv(
-; CHECK-INDUCTION-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
-; CHECK-INDUCTION-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-INDUCTION-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
-; CHECK-INDUCTION-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-INDUCTION:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-INDUCTION-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 16
-; CHECK-INDUCTION-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-INDUCTION:       [[VECTOR_PH]]:
-; CHECK-INDUCTION-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 15
-; CHECK-INDUCTION-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-INDUCTION-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-INDUCTION:       [[VECTOR_BODY]]:
-; CHECK-INDUCTION-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-INDUCTION-NEXT:    [[VEC_IND:%.*]] = phi <8 x i64> [ <i64 0, i64 1, i64 2, i64 3, i64 4, i64 5, i64 6, i64 7>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-INDUCTION-NEXT:    [[VEC_PHI:%.*]] = phi <8 x i64> [ splat (i64 -9223372036854775808), %[[VECTOR_PH]] ], [ [[TMP12:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-INDUCTION-NEXT:    [[VEC_PHI2:%.*]] = phi <8 x i64> [ splat (i64 -9223372036854775808), %[[VECTOR_PH]] ], [ [[TMP15:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-INDUCTION-NEXT:    [[STEP_ADD:%.*]] = add nuw <8 x i64> [[VEC_IND]], splat (i64 8)
-; CHECK-INDUCTION-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-INDUCTION-NEXT:    [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[TMP1]], i64 8
-; CHECK-INDUCTION-NEXT:    [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP1]], align 4
-; CHECK-INDUCTION-NEXT:    [[WIDE_LOAD3:%.*]] = load <8 x i32>, ptr [[TMP2]], align 4
-; CHECK-INDUCTION-NEXT:    [[TMP3:%.*]] = icmp eq <8 x i32> [[WIDE_LOAD]], splat (i32 11)
-; CHECK-INDUCTION-NEXT:    [[TMP22:%.*]] = icmp eq <8 x i32> [[WIDE_LOAD3]], splat (i32 11)
-; CHECK-INDUCTION-NEXT:    [[TMP12]] = select <8 x i1> [[TMP3]], <8 x i64> [[VEC_IND]], <8 x i64> [[VEC_PHI]]
-; CHECK-INDUCTION-NEXT:    [[TMP15]] = select <8 x i1> [[TMP22]], <8 x i64> [[STEP_ADD]], <8 x i64> [[VEC_PHI2]]
-; CHECK-INDUCTION-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 16
-; CHECK-INDUCTION-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <8 x i64> [[STEP_ADD]], splat (i64 8)
-; CHECK-INDUCTION-NEXT:    [[TMP4:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-INDUCTION-NEXT:    br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-INDUCTION:       [[MIDDLE_BLOCK]]:
-; CHECK-INDUCTION-NEXT:    [[RDX_MINMAX:%.*]] = call <8 x i64> @llvm.smax.v8i64(<8 x i64> [[TMP12]], <8 x i64> [[TMP15]])
-; CHECK-INDUCTION-NEXT:    [[TMP5:%.*]] = call i64 @llvm.vector.reduce.smax.v8i64(<8 x i64> [[RDX_MINMAX]])
-; CHECK-INDUCTION-NEXT:    [[TMP6:%.*]] = icmp ne i64 [[TMP5]], -9223372036854775808
-; CHECK-INDUCTION-NEXT:    [[TMP7:%.*]] = select i1 [[TMP6]], i64 [[TMP5]], i64 -1
-; CHECK-INDUCTION-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-INDUCTION-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-INDUCTION:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-INDUCTION-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-INDUCTION:       [[VEC_EPILOG_PH]]:
-; CHECK-INDUCTION-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-INDUCTION-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i64 [ [[TMP7]], %[[VEC_EPILOG_ITER_CHECK]] ], [ -1, %[[ITER_CHECK]] ], [ -1, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-INDUCTION-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[BC_MERGE_RDX]], -1
-; CHECK-INDUCTION-NEXT:    [[TMP9:%.*]] = select i1 [[TMP8]], i64 -9223372036854775808, i64 [[BC_MERGE_RDX]]
-; CHECK-INDUCTION-NEXT:    [[N_RND_UP:%.*]] = add i64 [[N]], 3
-; CHECK-INDUCTION-NEXT:    [[TMP10:%.*]] = and i64 [[N_RND_UP]], 3
-; CHECK-INDUCTION-NEXT:    [[N_VEC2:%.*]] = sub i64 [[N_RND_UP]], [[TMP10]]
-; CHECK-INDUCTION-NEXT:    [[TRIP_COUNT_MINUS_1:%.*]] = sub i64 [[N]], 1
-; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[TRIP_COUNT_MINUS_1]], i64 0
-; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLATINSERT5:%.*]] = insertelement <4 x i64> poison, i64 [[TMP9]], i64 0
-; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLAT6:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT5]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLATINSERT7:%.*]] = insertelement <4 x i64> poison, i64 [[VEC_EPILOG_RESUME_VAL]], i64 0
-; CHECK-INDUCTION-NEXT:    [[BROADCAST_SPLAT8:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT7]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-INDUCTION-NEXT:    [[INDUCTION:%.*]] = add nuw <4 x i64> [[BROADCAST_SPLAT8]], <i64 0, i64 1, i64 2, i64 3>
-; CHECK-INDUCTION-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-INDUCTION:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-INDUCTION-NEXT:    [[TMP11:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[TMP17:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-INDUCTION-NEXT:    [[VEC_IND10:%.*]] = phi <4 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT14:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-INDUCTION-NEXT:    [[VEC_PHI11:%.*]] = phi <4 x i64> [ [[BROADCAST_SPLAT6]], %[[VEC_EPILOG_PH]] ], [ [[TMP23:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-INDUCTION-NEXT:    [[VEC_IND12:%.*]] = phi <4 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT15:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-INDUCTION-NEXT:    [[TMP14:%.*]] = icmp ule <4 x i64> [[VEC_IND12]], [[BROADCAST_SPLAT]]
-; CHECK-INDUCTION-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[TMP11]]
-; CHECK-INDUCTION-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[TMP13]], <4 x i1> [[TMP14]], <4 x i32> poison)
-; CHECK-INDUCTION-NEXT:    [[TMP16:%.*]] = icmp eq <4 x i32> [[WIDE_MASKED_LOAD]], splat (i32 11)
-; CHECK-INDUCTION-NEXT:    [[TMP24:%.*]] = select <4 x i1> [[TMP14]], <4 x i1> [[TMP16]], <4 x i1> zeroinitializer
-; CHECK-INDUCTION-NEXT:    [[TMP23]] = select <4 x i1> [[TMP24]], <4 x i64> [[VEC_IND10]], <4 x i64> [[VEC_PHI11]]
-; CHECK-INDUCTION-NEXT:    [[TMP17]] = add i64 [[TMP11]], 4
-; CHECK-INDUCTION-NEXT:    [[VEC_IND_NEXT14]] = add nuw nsw <4 x i64> [[VEC_IND10]], splat (i64 4)
-; CHECK-INDUCTION-NEXT:    [[VEC_IND_NEXT15]] = add nuw <4 x i64> [[VEC_IND12]], splat (i64 4)
-; CHECK-INDUCTION-NEXT:    [[TMP18:%.*]] = icmp eq i64 [[TMP17]], [[N_VEC2]]
-; CHECK-INDUCTION-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-INDUCTION:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-INDUCTION-NEXT:    [[TMP19:%.*]] = call i64 @llvm.vector.reduce.smax.v4i64(<4 x i64> [[TMP23]])
-; CHECK-INDUCTION-NEXT:    [[TMP20:%.*]] = icmp ne i64 [[TMP19]], -9223372036854775808
-; CHECK-INDUCTION-NEXT:    [[TMP21:%.*]] = select i1 [[TMP20]], i64 [[TMP19]], i64 -1
-; CHECK-INDUCTION-NEXT:    br label %[[EXIT]]
-; CHECK-INDUCTION:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-INDUCTION-NEXT:    br label %[[LOOP:.*]]
-; CHECK-INDUCTION:       [[LOOP]]:
-; CHECK-INDUCTION-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-INDUCTION-NEXT:    [[RED:%.*]] = phi i64 [ -1, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SEL:%.*]], %[[LOOP]] ]
-; CHECK-INDUCTION-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
-; CHECK-INDUCTION-NEXT:    [[L:%.*]] = load i32, ptr [[GEP]], align 4
-; CHECK-INDUCTION-NEXT:    [[C:%.*]] = icmp eq i32 [[L]], 11
-; CHECK-INDUCTION-NEXT:    [[SEL]] = select i1 [[C]], i64 [[IV]], i64 [[RED]]
-; CHECK-INDUCTION-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-INDUCTION-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-INDUCTION-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
-; CHECK-INDUCTION:       [[EXIT]]:
-; CHECK-INDUCTION-NEXT:    [[SEL_LCSSA:%.*]] = phi i64 [ [[SEL]], %[[LOOP]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[TMP21]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-INDUCTION-NEXT:    ret i64 [[SEL_LCSSA]]
-;
-entry:
-  br label %loop
-
-loop:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-  %red = phi i64 [ -1, %entry ], [ %sel, %loop ]
-  %gep = getelementptr inbounds i32, ptr %A, i64 %iv
-  %l = load i32, ptr %gep, align 4
-  %c = icmp eq i32 %l, 11
-  %sel = select i1 %c, i64 %iv, i64 %red
-  %iv.next = add nuw nsw i64 %iv, 1
-  %ec = icmp eq i64 %iv.next, %n
-  br i1 %ec, label %exit, label %loop
-
-exit:
-  ret i64 %sel
-}
 
 define i64 @test_no_masked_interleave_support(i64 %y, i32 %n) {
 ; CHECK-INVALIDATE-INTERLEAVE-LABEL: Checking a loop in 'test_no_masked_interleave_support'
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll
index 8196fcdc6dc98..0a1eddf6e2804 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/partial-reduce-with-predicate.ll
@@ -1,7 +1,6 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "^scalar.ph" --version 6
 ; RUN: opt -passes=loop-vectorize -enable-epilogue-vectorization=false -S < %s | FileCheck %s --check-prefixes=CHECK
 ; RUN: opt -passes=loop-vectorize -enable-epilogue-vectorization=false -tail-folding-policy=must-fold-tail -S < %s | FileCheck %s --check-prefixes=CHECK-TAILFOLD
-; RUN: opt -passes=loop-vectorize -force-vector-width=16 -epilogue-vectorization-force-VF=8 -epilogue-tail-folding-policy=prefer-fold-tail -S < %s | FileCheck %s --check-prefixes=CHECK-TAILFOLD-EPILOGUE
 
 target triple = "aarch64-none-unknown-elf"
 
@@ -81,99 +80,6 @@ define i32 @pred_reduction(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP10]]
 ;
-; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @pred_reduction(
-; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[SRC:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE5:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP7]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP8]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP9]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE5]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP10]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE5]], [[PARTIAL_REDUCE]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP12]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX6:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI7:%.*]] = phi <2 x i32> [ [[TMP13]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX6]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD8:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP14]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD8]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP15]], <8 x i1> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX6]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD9:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP17]], <8 x i1> [[TMP16]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD9]] to <8 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = select <8 x i1> [[TMP16]], <8 x i32> [[TMP18]], <8 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE10]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI7]], <8 x i32> [[TMP19]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT11]] = add i64 [[INDEX6]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT11]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = xor i1 [[TMP20]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE10]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_1:%.*]], %[[FOR_INC]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VAL:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CONV:%.*]] = zext i8 [[VAL]] to i32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM]], [[CONV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1]] = phi i32 [ [[ADD]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[SUM_1]], %[[FOR_INC]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ], [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
-;
 entry:
   br label %for.body
 
@@ -278,99 +184,6 @@ define i32 @pred_reduction_sext(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP10]]
 ;
-; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @pred_reduction_sext(
-; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[SRC:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE5:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = sext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP7]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP8]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = sext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP9]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE5]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP10]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE5]], [[PARTIAL_REDUCE]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP12]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX6:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI7:%.*]] = phi <2 x i32> [ [[TMP13]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX6]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD8:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP14]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD8]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP15]], <8 x i1> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX6]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD9:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP17]], <8 x i1> [[TMP16]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = sext <8 x i8> [[WIDE_MASKED_LOAD9]] to <8 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = select <8 x i1> [[TMP16]], <8 x i32> [[TMP18]], <8 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE10]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI7]], <8 x i32> [[TMP19]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT11]] = add i64 [[INDEX6]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT11]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = xor i1 [[TMP20]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE10]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_1:%.*]], %[[FOR_INC]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VAL:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CONV:%.*]] = sext i8 [[VAL]] to i32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM]], [[CONV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1]] = phi i32 [ [[ADD]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[SUM_1]], %[[FOR_INC]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ], [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
-;
 entry:
   br label %for.body
 
@@ -487,115 +300,6 @@ define i32 @pred_reduction_dotprod(ptr %a, ptr %b, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP13]]
 ;
-; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @pred_reduction_dotprod(
-; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE7:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = getelementptr i8, ptr [[TMP7]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD5:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP7]], <16 x i1> [[TMP3]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD6:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP8]], <16 x i1> [[TMP4]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD5]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = mul nuw nsw <16 x i32> [[TMP9]], [[TMP10]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP11]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP12]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD6]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = mul nuw nsw <16 x i32> [[TMP13]], [[TMP14]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP15]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE7]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP16]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE7]], [[PARTIAL_REDUCE]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP18]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX8:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT14:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI9:%.*]] = phi <2 x i32> [ [[TMP19]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX8]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD10:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP20]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD10]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP21]], <8 x i1> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP23:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX8]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD11:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP23]], <8 x i1> [[TMP22]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP24:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX8]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD12:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP24]], <8 x i1> [[TMP22]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP25:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD11]] to <8 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP26:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD12]] to <8 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP27:%.*]] = mul nuw nsw <8 x i32> [[TMP25]], [[TMP26]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP28:%.*]] = select <8 x i1> [[TMP22]], <8 x i32> [[TMP27]], <8 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE13]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI9]], <8 x i32> [[TMP28]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT14]] = add i64 [[INDEX8]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT14]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP30:%.*]] = xor i1 [[TMP29]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP31:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE13]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_1:%.*]], %[[FOR_INC]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[LOAD_A:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX4:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[LOAD_B:%.*]] = load i8, ptr [[ARRAYIDX4]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXT_A:%.*]] = zext i8 [[LOAD_A]] to i32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXT_B:%.*]] = zext i8 [[LOAD_B]] to i32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MUL:%.*]] = mul nuw nsw i32 [[EXT_A]], [[EXT_B]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM]], [[MUL]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1]] = phi i32 [ [[ADD]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[SUM_1]], %[[FOR_INC]] ], [ [[TMP18]], %[[MIDDLE_BLOCK]] ], [ [[TMP31]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
-;
 entry:
   br label %for.body
 
@@ -718,116 +422,6 @@ define i32 @pred_sub_reduction(ptr %a, ptr %b, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP14]]
 ;
-; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @pred_sub_reduction(
-; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE7:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = getelementptr i8, ptr [[TMP7]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD5:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP7]], <16 x i1> [[TMP3]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD6:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP8]], <16 x i1> [[TMP4]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD5]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = mul nuw nsw <16 x i32> [[TMP9]], [[TMP10]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP11]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP12]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD6]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = mul nuw nsw <16 x i32> [[TMP13]], [[TMP14]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP15]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE7]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP16]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE7]], [[PARTIAL_REDUCE]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = sub i32 0, [[TMP18]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP19]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX8:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT14:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI9:%.*]] = phi <2 x i32> [ zeroinitializer, %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE13:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX8]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD10:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP20]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD10]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP21]], <8 x i1> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP23:%.*]] = getelementptr i8, ptr [[A]], i64 [[INDEX8]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD11:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP23]], <8 x i1> [[TMP22]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP24:%.*]] = getelementptr i8, ptr [[B]], i64 [[INDEX8]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD12:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP24]], <8 x i1> [[TMP22]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP25:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD11]] to <8 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP26:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD12]] to <8 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP27:%.*]] = mul nuw nsw <8 x i32> [[TMP25]], [[TMP26]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP28:%.*]] = select <8 x i1> [[TMP22]], <8 x i32> [[TMP27]], <8 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE13]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI9]], <8 x i32> [[TMP28]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT14]] = add i64 [[INDEX8]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT14]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP30:%.*]] = xor i1 [[TMP29]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP31:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE13]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP32:%.*]] = sub i32 [[BC_MERGE_RDX]], [[TMP31]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_1:%.*]], %[[FOR_INC]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[A]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[LOAD_A:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX4:%.*]] = getelementptr inbounds nuw i8, ptr [[B]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[LOAD_B:%.*]] = load i8, ptr [[ARRAYIDX4]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXT_A:%.*]] = zext i8 [[LOAD_A]] to i32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXT_B:%.*]] = zext i8 [[LOAD_B]] to i32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MUL:%.*]] = mul nuw nsw i32 [[EXT_A]], [[EXT_B]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUB:%.*]] = sub nsw i32 [[SUM]], [[MUL]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1]] = phi i32 [ [[SUB]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[SUM_1]], %[[FOR_INC]] ], [ [[TMP19]], %[[MIDDLE_BLOCK]] ], [ [[TMP32]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
-;
 entry:
   br label %for.body
 
@@ -949,116 +543,6 @@ define i32 @chained_pred_reduction(ptr %src, ptr noalias %src_b, ptr %cond, i64
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP13]]
 ;
-; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @chained_pred_reduction(
-; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[SRC:%.*]], ptr noalias [[SRC_B:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE8:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE9:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP7]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE:%.*]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP8]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP9]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE5:%.*]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP10]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP11]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD6:%.*]] = load <16 x i8>, ptr [[TMP11]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD7:%.*]] = load <16 x i8>, ptr [[TMP12]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = zext <16 x i8> [[WIDE_LOAD6]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE8]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[PARTIAL_REDUCE]], <16 x i32> [[TMP13]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = zext <16 x i8> [[WIDE_LOAD7]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE9]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[PARTIAL_REDUCE5]], <16 x i32> [[TMP14]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE9]], [[PARTIAL_REDUCE8]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP16]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT17:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI11:%.*]] = phi <2 x i32> [ [[TMP17]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE16:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX10]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD12:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP18]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD12]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP19]], <8 x i1> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX10]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD13:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP21]], <8 x i1> [[TMP20]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD13]] to <8 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP23:%.*]] = select <8 x i1> [[TMP20]], <8 x i32> [[TMP22]], <8 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE14:%.*]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI11]], <8 x i32> [[TMP23]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP24:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[INDEX10]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD15:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP24]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP25:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD15]] to <8 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP26:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP25]], <8 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE16]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[PARTIAL_REDUCE14]], <8 x i32> [[TMP26]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT17]] = add i64 [[INDEX10]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT17]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP27:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP28:%.*]] = xor i1 [[TMP27]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP28]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE16]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_2:%.*]], %[[FOR_INC]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VAL:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CONV:%.*]] = zext i8 [[VAL]] to i32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM]], [[CONV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1:%.*]] = phi i32 [ [[ADD]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[B_GEP:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BVAL:%.*]] = load i8, ptr [[B_GEP]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BCONV:%.*]] = zext i8 [[BVAL]] to i32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2]] = add nsw i32 [[SUM_1]], [[BCONV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2_LCSSA:%.*]] = phi i32 [ [[SUM_2]], %[[FOR_INC]] ], [ [[TMP16]], %[[MIDDLE_BLOCK]] ], [ [[TMP29]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_2_LCSSA]]
-;
 entry:
   br label %for.body
 
@@ -1180,116 +664,6 @@ define i32 @reduction_before_pred(ptr %src, ptr noalias %src_b, ptr %cond, i64 %
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP12]]
 ;
-; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @reduction_before_pred(
-; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[SRC:%.*]], ptr noalias [[SRC_B:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE8:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE9:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = zext <16 x i8> [[WIDE_LOAD]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE:%.*]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP3]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = zext <16 x i8> [[WIDE_LOAD3]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE4:%.*]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP4]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP5]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD5:%.*]] = load <16 x i8>, ptr [[TMP5]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD6:%.*]] = load <16 x i8>, ptr [[TMP6]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD5]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD6]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = getelementptr i8, ptr [[TMP9]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP9]], <16 x i1> [[TMP7]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD7:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP10]], <16 x i1> [[TMP8]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = select <16 x i1> [[TMP7]], <16 x i32> [[TMP11]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE8]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[PARTIAL_REDUCE]], <16 x i32> [[TMP12]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD7]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = select <16 x i1> [[TMP8]], <16 x i32> [[TMP13]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE9]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[PARTIAL_REDUCE4]], <16 x i32> [[TMP14]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP15]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE9]], [[PARTIAL_REDUCE8]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP16]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX10:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT17:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI11:%.*]] = phi <2 x i32> [ [[TMP17]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE16:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[INDEX10]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD12:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP18]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD12]] to <8 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> [[TMP19]], <8 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE13:%.*]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI11]], <8 x i32> [[TMP20]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX10]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD14:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP21]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD14]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP23:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP22]], <8 x i1> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP24:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX10]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD15:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP24]], <8 x i1> [[TMP23]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP25:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD15]] to <8 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP26:%.*]] = select <8 x i1> [[TMP23]], <8 x i32> [[TMP25]], <8 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE16]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[PARTIAL_REDUCE13]], <8 x i32> [[TMP26]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT17]] = add i64 [[INDEX10]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT17]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP27:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP28:%.*]] = xor i1 [[TMP27]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP28]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP29:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE16]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_2:%.*]], %[[FOR_INC]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[B_GEP:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC_B]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BVAL:%.*]] = load i8, ptr [[B_GEP]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BCONV:%.*]] = zext i8 [[BVAL]] to i32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1:%.*]] = add nsw i32 [[SUM]], [[BCONV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VAL:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CONV:%.*]] = zext i8 [[VAL]] to i32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM_1]], [[CONV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2]] = phi i32 [ [[SUM_1]], %[[FOR_BODY]] ], [ [[ADD]], %[[IF_THEN]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_2_LCSSA:%.*]] = phi i32 [ [[SUM_2]], %[[FOR_INC]] ], [ [[TMP16]], %[[MIDDLE_BLOCK]] ], [ [[TMP29]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_2_LCSSA]]
-;
 entry:
   br label %for.body
 
@@ -1399,106 +773,13 @@ define i32 @pred_reduction_incoming_1(ptr %src, ptr %cond, i64 %N) #0 {
 ; CHECK-TAILFOLD-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 16 x i1> @llvm.get.active.lane.mask.nxv16i1.i64(i64 [[INDEX_NEXT]], i64 [[N]])
 ; CHECK-TAILFOLD-NEXT:    [[TMP12:%.*]] = extractelement <vscale x 16 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-TAILFOLD-NEXT:    [[TMP13:%.*]] = xor i1 [[TMP12]], true
-; CHECK-TAILFOLD-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-TAILFOLD-NEXT:    br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
 ; CHECK-TAILFOLD:       [[MIDDLE_BLOCK]]:
 ; CHECK-TAILFOLD-NEXT:    [[TMP14:%.*]] = call i32 @llvm.vector.reduce.add.nxv4i32(<vscale x 4 x i32> [[PARTIAL_REDUCE]])
 ; CHECK-TAILFOLD-NEXT:    br label %[[EXIT:.*]]
 ; CHECK-TAILFOLD:       [[EXIT]]:
 ; CHECK-TAILFOLD-NEXT:    ret i32 [[TMP14]]
 ;
-; CHECK-TAILFOLD-EPILOGUE-LABEL: define i32 @pred_reduction_incoming_1(
-; CHECK-TAILFOLD-EPILOGUE-SAME: ptr [[SRC:%.*]], ptr [[COND:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-TAILFOLD-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PARTIAL_REDUCE5:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP1]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i8>, ptr [[TMP1]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_LOAD3:%.*]] = load <16 x i8>, ptr [[TMP2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP4:%.*]] = icmp ne <16 x i8> [[WIDE_LOAD3]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[TMP5]], i64 16
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP5]], <16 x i1> [[TMP3]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD4:%.*]] = call <16 x i8> @llvm.masked.load.v16i8.p0(ptr align 1 [[TMP6]], <16 x i1> [[TMP4]], <16 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP7:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP8:%.*]] = select <16 x i1> [[TMP3]], <16 x i32> [[TMP7]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI]], <16 x i32> [[TMP8]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP9:%.*]] = zext <16 x i8> [[WIDE_MASKED_LOAD4]] to <16 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP10:%.*]] = select <16 x i1> [[TMP4]], <16 x i32> [[TMP9]], <16 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE5]] = call <4 x i32> @llvm.vector.partial.reduce.add.v4i32.v16i32(<4 x i32> [[VEC_PHI2]], <16 x i32> [[TMP10]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BIN_RDX:%.*]] = add <4 x i32> [[PARTIAL_REDUCE5]], [[PARTIAL_REDUCE]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP12:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[BIN_RDX]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP12]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP13:%.*]] = insertelement <2 x i32> zeroinitializer, i32 [[BC_MERGE_RDX]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX6:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT11:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VEC_PHI7:%.*]] = phi <2 x i32> [ [[TMP13]], %[[VEC_EPILOG_PH]] ], [ [[PARTIAL_REDUCE10:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP14:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[INDEX6]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD8:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP14]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP15:%.*]] = icmp ne <8 x i8> [[WIDE_MASKED_LOAD8]], zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP16:%.*]] = select <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> [[TMP15]], <8 x i1> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP17:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[INDEX6]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD9:%.*]] = call <8 x i8> @llvm.masked.load.v8i8.p0(ptr align 1 [[TMP17]], <8 x i1> [[TMP16]], <8 x i8> poison)
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP18:%.*]] = zext <8 x i8> [[WIDE_MASKED_LOAD9]] to <8 x i32>
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP19:%.*]] = select <8 x i1> [[TMP16]], <8 x i32> [[TMP18]], <8 x i32> zeroinitializer
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[PARTIAL_REDUCE10]] = call <2 x i32> @llvm.vector.partial.reduce.add.v2i32.v8i32(<2 x i32> [[VEC_PHI7]], <8 x i32> [[TMP19]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[INDEX_NEXT11]] = add i64 [[INDEX6]], 8
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT11]], i64 [[N]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP20:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP21:%.*]] = xor i1 [[TMP20]], true
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TMP21]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TMP22:%.*]] = call i32 @llvm.vector.reduce.add.v2i32(<2 x i32> [[PARTIAL_REDUCE10]])
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[EXIT]]
-; CHECK-TAILFOLD-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_BODY:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[IF_THEN:.*]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX2:%.*]] = getelementptr inbounds nuw i8, ptr [[SRC]], i64 [[IV:%.*]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[VAL:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[CONV:%.*]] = zext i8 [[VAL]] to i32
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ADD:%.*]] = add nsw i32 [[SUM:%.*]], [[CONV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br label %[[FOR_INC:.*]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_BODY]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM]] = phi i32 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[SUM_1:%.*]], %[[FOR_INC]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw i8, ptr [[COND]], i64 [[IV]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[C:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[TOBOOL_NOT:%.*]] = icmp eq i8 [[C]], 0
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[TOBOOL_NOT]], label %[[FOR_INC]], label %[[IF_THEN]]
-; CHECK-TAILFOLD-EPILOGUE:       [[FOR_INC]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1]] = phi i32 [ [[ADD]], %[[IF_THEN]] ], [ [[SUM]], %[[FOR_BODY]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    br i1 [[EXITCOND_NOT]], label %[[EXIT]], label %[[FOR_BODY]]
-; CHECK-TAILFOLD-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    [[SUM_1_LCSSA:%.*]] = phi i32 [ [[SUM_1]], %[[FOR_INC]] ], [ [[TMP12]], %[[MIDDLE_BLOCK]] ], [ [[TMP22]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLD-EPILOGUE-NEXT:    ret i32 [[SUM_1_LCSSA]]
-;
 entry:
   br label %for.body
 
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll
index e1086979859cf..1ac73e9d6bf74 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-fixed-order-recurrences.ll
@@ -1,7 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
 ; RUN: opt -passes=loop-vectorize -force-vector-width=8 -enable-epilogue-vectorization -epilogue-vectorization-force-VF=4 -S %s | FileCheck %s
-; RUN: opt -passes=loop-vectorize -force-vector-width=8 -epilogue-vectorization-force-VF=4 -epilogue-tail-folding-policy=prefer-fold-tail \
-; RUN: -force-target-supports-masked-memory-ops -S %s | FileCheck %s --check-prefix=CHECK-TAILFOLDED-EPILOGUE
 
 
 define void @dead_for(ptr %a, i64 %N) {
@@ -68,74 +66,6 @@ define void @dead_for(ptr %a, i64 %N) {
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
-; CHECK-TAILFOLDED-EPILOGUE-LABEL: define void @dead_for(
-; CHECK-TAILFOLDED-EPILOGUE-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_PH]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 7
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_BODY]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <8 x i64>, ptr [[TMP1]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP2:%.*]] = add <8 x i64> [[WIDE_LOAD]], splat (i64 10)
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store <8 x i64> [[TMP2]], ptr [[TMP1]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[MIDDLE_BLOCK]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP4:%.*]] = extractelement <8 x i64> [[WIDE_LOAD]], i64 7
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_PH]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_RND_UP:%.*]] = add i64 [[N]], 3
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP5:%.*]] = and i64 [[N_RND_UP]], 3
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_VEC2:%.*]] = sub i64 [[N_RND_UP]], [[TMP5]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TRIP_COUNT_MINUS_1:%.*]] = sub i64 [[N]], 1
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[TRIP_COUNT_MINUS_1]], i64 0
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <4 x i64> poison, i64 [[VEC_EPILOG_RESUME_VAL]], i64 0
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT3]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDUCTION:%.*]] = add nuw <4 x i64> [[BROADCAST_SPLAT4]], <i64 0, i64 1, i64 2, i64 3>
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX3:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP6:%.*]] = icmp ule <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP7:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX3]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <4 x i64> @llvm.masked.load.v4i64.p0(ptr align 4 [[TMP7]], <4 x i1> [[TMP6]], <4 x i64> poison)
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP8:%.*]] = add <4 x i64> [[WIDE_MASKED_LOAD]], splat (i64 10)
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    call void @llvm.masked.store.v4i64.p0(<4 x i64> [[TMP8]], ptr align 4 [[TMP7]], <4 x i1> [[TMP6]])
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX_NEXT4]] = add i64 [[INDEX3]], 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_IND_NEXT]] = add nuw <4 x i64> [[VEC_IND]], splat (i64 4)
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT4]], [[N_VEC2]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[TMP9]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[EXIT]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[LOOP:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[LOOP]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR:%.*]] = phi i64 [ 99, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[L:%.*]], %[[LOOP]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[GEP:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L]] = load i64, ptr [[GEP]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[ADD:%.*]] = add i64 [[L]], 10
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store i64 [[ADD]], ptr [[GEP]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    ret void
-;
 entry:
   br label %loop
 
@@ -199,50 +129,6 @@ define i64 @for_phi_used_in_loop_and_live_out(ptr %a, i64 %N) {
 ; CHECK-NEXT:    [[RES:%.*]] = add i64 [[L_LCSSA]], [[FOR_LCSSA]]
 ; CHECK-NEXT:    ret i64 [[RES]]
 ;
-; CHECK-TAILFOLDED-EPILOGUE-LABEL: define i64 @for_phi_used_in_loop_and_live_out(
-; CHECK-TAILFOLDED-EPILOGUE-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:  [[ENTRY:.*]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_PH]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 7
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_BODY]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR:%.*]] = phi <8 x i64> [ <i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 poison, i64 99>, %[[VECTOR_PH]] ], [ [[WIDE_LOAD:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[WIDE_LOAD]] = load <8 x i64>, ptr [[TMP1]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP2:%.*]] = shufflevector <8 x i64> [[VECTOR_RECUR]], <8 x i64> [[WIDE_LOAD]], <8 x i32> <i32 7, i32 8, i32 9, i32 10, i32 11, i32 12, i32 13, i32 14>
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store <8 x i64> [[TMP2]], ptr [[TMP1]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[MIDDLE_BLOCK]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR_EXTRACT_FOR_PHI:%.*]] = extractelement <8 x i64> [[WIDE_LOAD]], i64 6
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <8 x i64> [[WIDE_LOAD]], i64 7
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[SCALAR_PH]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[SCALAR_RECUR_INIT:%.*]] = phi i64 [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ], [ 99, %[[ENTRY]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[LOOP:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[LOOP]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR:%.*]] = phi i64 [ [[SCALAR_RECUR_INIT]], %[[SCALAR_PH]] ], [ [[L:%.*]], %[[LOOP]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[GEP:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L]] = load i64, ptr [[GEP]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[ADD:%.*]] = add i64 [[L]], 10
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store i64 [[FOR]], ptr [[GEP]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR_LCSSA:%.*]] = phi i64 [ [[FOR]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L_LCSSA:%.*]] = phi i64 [ [[L]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[RES:%.*]] = add i64 [[L_LCSSA]], [[FOR_LCSSA]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    ret i64 [[RES]]
-;
 entry:
   br label %loop
 
@@ -305,7 +191,7 @@ define i64 @for_phi_not_used_in_loop_and_live_out(ptr %a, i64 %N) {
 ; CHECK-NEXT:    store <4 x i64> [[TMP4]], ptr [[GEP]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT6]] = add nuw i64 [[IV]], 4
 ; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT6]], [[N_VEC3]]
-; CHECK-NEXT:    br i1 [[TMP5]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[LOOP]]
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[LOOP]], !llvm.loop [[LOOP9:![0-9]+]]
 ; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-NEXT:    [[VECTOR_RECUR_EXTRACT_FOR_PHI7:%.*]] = extractelement <4 x i64> [[WIDE_LOAD5]], i64 2
 ; CHECK-NEXT:    [[VECTOR_RECUR_EXTRACT8:%.*]] = extractelement <4 x i64> [[WIDE_LOAD5]], i64 3
@@ -331,87 +217,6 @@ define i64 @for_phi_not_used_in_loop_and_live_out(ptr %a, i64 %N) {
 ; CHECK-NEXT:    [[RES:%.*]] = add i64 [[L_LCSSA]], [[FOR_LCSSA]]
 ; CHECK-NEXT:    ret i64 [[RES]]
 ;
-; CHECK-TAILFOLDED-EPILOGUE-LABEL: define i64 @for_phi_not_used_in_loop_and_live_out(
-; CHECK-TAILFOLDED-EPILOGUE-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_PH]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 7
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VECTOR_BODY]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[WIDE_LOAD:%.*]] = load <8 x i64>, ptr [[TMP1]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP2:%.*]] = add <8 x i64> [[WIDE_LOAD]], splat (i64 10)
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store <8 x i64> [[TMP2]], ptr [[TMP1]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[MIDDLE_BLOCK]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR_EXTRACT_FOR_PHI:%.*]] = extractelement <8 x i64> [[WIDE_LOAD]], i64 6
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <8 x i64> [[WIDE_LOAD]], i64 7
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_PH]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[SCALAR_RECUR_INIT:%.*]] = phi i64 [ [[VECTOR_RECUR_EXTRACT]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 99, %[[ITER_CHECK]] ], [ 99, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_RND_UP:%.*]] = add i64 [[N]], 3
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP4:%.*]] = and i64 [[N_RND_UP]], 3
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[N_VEC2:%.*]] = sub i64 [[N_RND_UP]], [[TMP4]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TRIP_COUNT_MINUS_1:%.*]] = sub i64 [[N]], 1
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i64> poison, i64 [[TRIP_COUNT_MINUS_1]], i64 0
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <4 x i64> poison, i64 [[VEC_EPILOG_RESUME_VAL]], i64 0
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <4 x i64> [[BROADCAST_SPLATINSERT3]], <4 x i64> poison, <4 x i32> zeroinitializer
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDUCTION:%.*]] = add nuw <4 x i64> [[BROADCAST_SPLAT4]], <i64 0, i64 1, i64 2, i64 3>
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR_INIT:%.*]] = insertelement <4 x i64> poison, i64 [[SCALAR_RECUR_INIT]], i32 3
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX3:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VECTOR_RECUR:%.*]] = phi <4 x i64> [ [[VECTOR_RECUR_INIT]], %[[VEC_EPILOG_PH]] ], [ [[WIDE_MASKED_LOAD:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_IND:%.*]] = phi <4 x i64> [ [[INDUCTION]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP5:%.*]] = icmp ule <4 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP6:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[INDEX3]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[WIDE_MASKED_LOAD]] = call <4 x i64> @llvm.masked.load.v4i64.p0(ptr align 4 [[TMP6]], <4 x i1> [[TMP5]], <4 x i64> poison)
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP7:%.*]] = add <4 x i64> [[WIDE_MASKED_LOAD]], splat (i64 10)
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    call void @llvm.masked.store.v4i64.p0(<4 x i64> [[TMP7]], ptr align 4 [[TMP6]], <4 x i1> [[TMP5]])
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[INDEX_NEXT4]] = add i64 [[INDEX3]], 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[VEC_IND_NEXT]] = add nuw <4 x i64> [[VEC_IND]], splat (i64 4)
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT4]], [[N_VEC2]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[TMP8]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP9:%.*]] = shufflevector <4 x i64> [[VECTOR_RECUR]], <4 x i64> [[WIDE_MASKED_LOAD]], <4 x i32> <i32 3, i32 4, i32 5, i32 6>
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP10:%.*]] = xor <4 x i1> [[TMP5]], splat (i1 true)
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP10]], i1 false)
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP11:%.*]] = extractelement <4 x i64> [[TMP9]], i64 [[LAST_ACTIVE_LANE]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[TMP12:%.*]] = extractelement <4 x i64> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[EXIT]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[VEC_EPILOG_SCALAR_PH]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br label %[[LOOP:.*]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[LOOP]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR:%.*]] = phi i64 [ 99, %[[VEC_EPILOG_SCALAR_PH]] ], [ [[L:%.*]], %[[LOOP]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[GEP:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L]] = load i64, ptr [[GEP]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[ADD:%.*]] = add i64 [[L]], 10
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    store i64 [[ADD]], ptr [[GEP]], align 4
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    br i1 [[EC]], label %[[EXIT]], label %[[LOOP]]
-; CHECK-TAILFOLDED-EPILOGUE:       [[EXIT]]:
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[FOR_LCSSA:%.*]] = phi i64 [ [[FOR]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT_FOR_PHI]], %[[MIDDLE_BLOCK]] ], [ [[TMP11]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[L_LCSSA:%.*]] = phi i64 [ [[L]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ], [ [[TMP12]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    [[RES:%.*]] = add i64 [[L_LCSSA]], [[FOR_LCSSA]]
-; CHECK-TAILFOLDED-EPILOGUE-NEXT:    ret i64 [[RES]]
-;
 entry:
   br label %loop
 

>From 000733b8169d7bf56f24bce6bd963fe0188f2115 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Sat, 19 Sep 2026 14:29:13 +0000
Subject: [PATCH 19/25] add unittest for VPWidenCanonicalIVRecipe and
 CanoicalIV

---
 .../Transforms/Vectorize/VPlanTest.cpp         | 18 ++++++++++++++++++
 1 file changed, 18 insertions(+)

diff --git a/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp b/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
index ffc3a0d303474..5a62cd122265d 100644
--- a/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
+++ b/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
@@ -1760,6 +1760,24 @@ TEST_F(VPRecipeTest, CastVPReductionEVLRecipeToVPUser) {
   VPReductionEVLRecipe EVLRecipe(Recipe, *EVL, CondOp);
   checkVPRecipeCastImpl<VPReductionEVLRecipe, VPUser>(&EVLRecipe);
 }
+
+TEST_F(VPRecipeTest, CastVPWidenCanonicalIVRecipeToVPUser) {
+  VPlan &Plan = getPlan();
+  VPBasicBlock *Preheader = Plan.getEntry();
+  VPBasicBlock *Header = Plan.createVPBasicBlock("header");
+  VPBasicBlock *Latch = Plan.createVPBasicBlock("latch");
+  VPRegionBlock *Region = Plan.createLoopRegion(Type::getInt32Ty(C), DebugLoc(),
+                                                "loop", Header, Latch);
+  VPBlockUtils::connectBlocks(Header, Latch);
+  VPBlockUtils::connectBlocks(Preheader, Region);
+  VPBlockUtils::connectBlocks(Region, Plan.getScalarHeader());
+
+  VPRegionValue *CanIV = Region->getCanonicalIV();
+  VPWidenCanonicalIVRecipe Recipe(CanIV);
+
+  EXPECT_EQ(CanIV, Recipe.getCanonicalIV());
+  checkVPRecipeCastImpl<VPWidenCanonicalIVRecipe, VPUser>(&Recipe);
+}
 } // namespace
 
 struct VPDoubleValueDef : public VPRecipeBase {

>From 1281fe2dd6eb7626707232cc371058b311e38769 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Mon, 21 Sep 2026 16:10:56 +0000
Subject: [PATCH 20/25] update sve-widen-phi.ll after rebase

---
 llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll | 3 +--
 1 file changed, 1 insertion(+), 2 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
index 1f36d26c59fda..ab86f398ebd2e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
@@ -110,8 +110,7 @@ define void @widen_ptr_phi_unrolled(ptr noalias nocapture %a, ptr noalias nocapt
 ; CHECK-EPI-TF:       vector.body:
 ; CHECK-EPI-TF-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
 ; CHECK-EPI-TF-NEXT:    [[TMP6:%.*]] = shl i64 [[INDEX]], 3
-; CHECK-EPI-TF-NEXT:    [[TMP7:%.*]] = add i64 [[TMP3]], 0
-; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = mul i64 [[TMP7]], 8
+; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = mul i64 [[TMP3]], 8
 ; CHECK-EPI-TF-NEXT:    [[TMP9:%.*]] = add i64 [[TMP6]], [[TMP8]]
 ; CHECK-EPI-TF-NEXT:    [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP6]]
 ; CHECK-EPI-TF-NEXT:    [[NEXT_GEP2:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP9]]

>From 1d2f09bead7297b6b9c6f4812bf7c1a36bf5ba92 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Mon, 21 Sep 2026 16:37:24 +0000
Subject: [PATCH 21/25] skip jumpping to VEC_EPILOG_SCALAR_PH instead of
 VEC_EPILOG_PH when we have mem.check or scev.check so that vectorization can
 be skipped when mem/scev checks fail.

---
 .../Transforms/Vectorize/LoopVectorize.cpp    |  15 +-
 .../AArch64/fold-epilogue-tail.ll             | 181 +++++++++++++++++-
 2 files changed, 187 insertions(+), 9 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index fa2b0f1e59038..e939b164e638d 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7807,10 +7807,15 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
   // the epilogue plan.
   BasicBlock *EpilogueIterationCountCheck =
       cast<VPIRBasicBlock>(EpiPlan.getEntry())->getIRBasicBlock();
-  if (IsEpilogueTfEnabled) {
-    // With a tail-folded epilogue there is no scalar remainder to bail
-    // to, even a trip count too small for the epilogue VF is handled safely by
-    // the masked epilogue vector loop, so skip straight to its preheader.
+  bool JumpToScalarPH =
+      (!IsEpilogueTfEnabled || SCEVCheckBlock || MemCheckBlock);
+  // With tail-folding, the epilogue vector loop itself safely handles any
+  // trip count (including one smaller than the epilogue VF), so there's no
+  // need for a scalar remainder and we can jump straight to the epilogue
+  // preheader. The exception is when a SCEV or memory runtime check is
+  // present: those can fail at runtime regardless of tail-folding, so the
+  // scalar loop must still be kept as a fallback.
+  if (!JumpToScalarPH) {
     assert(is_contained(successors(EpilogueIterationCountCheck), ScalarPH) &&
            "expected iter.check to branch to the scalar preheader");
     ScalarPH->removePredecessor(EpilogueIterationCountCheck,
@@ -7843,7 +7848,7 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
     // TODO: revisit for reduction phis, whose resume value on this bypass
     // edge may need dedicated handling rather than reusing the value already
     // present here.
-    if (IsEpilogueTfEnabled)
+    if (!JumpToScalarPH)
       Phi->addIncoming(
           Phi->getIncomingValueForBlock(MainLoopIterationCountCheck),
           EpilogueIterationCountCheck);
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 173297610c9dd..2679cfcc7aedf 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -305,6 +305,179 @@ for.end:
   ret i32 %load
 }
 
+define void @stride_copy(ptr noalias nofree noundef writeonly captures(none) %dst, ptr noalias nofree noundef readonly captures(none) %src, i64 noundef %n, i64 noundef %stride) local_unnamed_addr #0 {
+; CHECK-LABEL: define void @stride_copy(
+; CHECK-SAME: ptr noalias nofree noundef writeonly captures(none) [[DST:%.*]], ptr noalias nofree noundef readonly captures(none) [[SRC:%.*]], i64 noundef [[N:%.*]], i64 noundef [[STRIDE:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[CMP5_NOT:%.*]] = icmp eq i64 [[N]], 0
+; CHECK-NEXT:    br i1 [[CMP5_NOT]], label %[[FOR_COND_CLEANUP:.*]], label %[[ITER_CHECK:.*]]
+; CHECK:       [[ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK:       [[VECTOR_SCEVCHECK]]:
+; CHECK-NEXT:    [[IDENT_CHECK:%.*]] = icmp ne i64 [[STRIDE]], 1
+; CHECK-NEXT:    br i1 [[IDENT_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP1]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP3]], i64 16
+; CHECK-NEXT:    store <16 x i32> [[WIDE_LOAD]], ptr [[TMP3]], align 4
+; CHECK-NEXT:    store <16 x i32> [[WIDE_LOAD2]], ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
+; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP5]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[FOR_COND_CLEANUP_LOOPEXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX3:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[INDEX3]]
+; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 4 [[TMP6]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[INDEX3]]
+; CHECK-NEXT:    call void @llvm.masked.store.v8i32.p0(<8 x i32> [[WIDE_MASKED_LOAD]], ptr align 4 [[TMP7]], <8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-NEXT:    [[INDEX_NEXT4]] = add i64 [[INDEX3]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT4]], i64 [[N]])
+; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-NEXT:    [[TMP9:%.*]] = xor i1 [[TMP8]], true
+; CHECK-NEXT:    br i1 [[TMP9]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[FOR_COND_CLEANUP_LOOPEXIT]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_COND_CLEANUP_LOOPEXIT]]:
+; CHECK-NEXT:    br label %[[FOR_COND_CLEANUP]]
+; CHECK:       [[FOR_COND_CLEANUP]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[I_06:%.*]] = phi i64 [ [[INC:%.*]], %[[FOR_BODY]] ], [ 0, %[[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-NEXT:    [[MUL:%.*]] = mul i64 [[I_06]], [[STRIDE]]
+; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[MUL]]
+; CHECK-NEXT:    [[TMP10:%.*]] = load i32, ptr [[ARRAYIDX]], align 4
+; CHECK-NEXT:    [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[I_06]]
+; CHECK-NEXT:    store i32 [[TMP10]], ptr [[ARRAYIDX1]], align 4
+; CHECK-NEXT:    [[INC]] = add nuw i64 [[I_06]], 1
+; CHECK-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INC]], [[N]]
+; CHECK-NEXT:    br i1 [[EXITCOND_NOT]], label %[[FOR_COND_CLEANUP_LOOPEXIT]], label %[[FOR_BODY]]
+;
+; CHECK-VS-LABEL: define void @stride_copy(
+; CHECK-VS-SAME: ptr noalias nofree noundef writeonly captures(none) [[DST:%.*]], ptr noalias nofree noundef readonly captures(none) [[SRC:%.*]], i64 noundef [[N:%.*]], i64 noundef [[STRIDE:%.*]]) local_unnamed_addr #[[ATTR0]] {
+; CHECK-VS-NEXT:  [[ENTRY:.*:]]
+; CHECK-VS-NEXT:    [[CMP5_NOT:%.*]] = icmp eq i64 [[N]], 0
+; CHECK-VS-NEXT:    br i1 [[CMP5_NOT]], label %[[FOR_COND_CLEANUP:.*]], label %[[ITER_CHECK:.*]]
+; CHECK-VS:       [[ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK-VS:       [[VECTOR_SCEVCHECK]]:
+; CHECK-VS-NEXT:    [[IDENT_CHECK:%.*]] = icmp ne i64 [[STRIDE]], 1
+; CHECK-VS-NEXT:    br i1 [[IDENT_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK-VS:       [[VECTOR_PH]]:
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK-VS:       [[VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP4]], i64 [[TMP3]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP4]], align 4
+; CHECK-VS-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 16 x i32>, ptr [[TMP5]], align 4
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP6]], i64 [[TMP3]]
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[WIDE_LOAD]], ptr [[TMP6]], align 4
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[WIDE_LOAD2]], ptr [[TMP7]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS:       [[MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[FOR_COND_CLEANUP_LOOPEXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
+; CHECK-VS:       [[VEC_EPILOG_PH]]:
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = shl nuw i64 [[TMP9]], 3
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
+; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-VS-NEXT:    [[INDEX3:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[INDEX3]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 4 [[TMP11]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[INDEX3]]
+; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[WIDE_MASKED_LOAD]], ptr align 4 [[TMP12]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT4]] = add i64 [[INDEX3]], [[TMP10]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT4]], i64 [[N]])
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = xor i1 [[TMP13]], true
+; CHECK-VS-NEXT:    br i1 [[TMP14]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-VS-NEXT:    br label %[[FOR_COND_CLEANUP_LOOPEXIT]]
+; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-VS-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK-VS:       [[FOR_COND_CLEANUP_LOOPEXIT]]:
+; CHECK-VS-NEXT:    br label %[[FOR_COND_CLEANUP]]
+; CHECK-VS:       [[FOR_COND_CLEANUP]]:
+; CHECK-VS-NEXT:    ret void
+; CHECK-VS:       [[FOR_BODY]]:
+; CHECK-VS-NEXT:    [[I_06:%.*]] = phi i64 [ [[INC:%.*]], %[[FOR_BODY]] ], [ 0, %[[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-VS-NEXT:    [[MUL:%.*]] = mul i64 [[I_06]], [[STRIDE]]
+; CHECK-VS-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[MUL]]
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = load i32, ptr [[ARRAYIDX]], align 4
+; CHECK-VS-NEXT:    [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[I_06]]
+; CHECK-VS-NEXT:    store i32 [[TMP15]], ptr [[ARRAYIDX1]], align 4
+; CHECK-VS-NEXT:    [[INC]] = add nuw i64 [[I_06]], 1
+; CHECK-VS-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INC]], [[N]]
+; CHECK-VS-NEXT:    br i1 [[EXITCOND_NOT]], label %[[FOR_COND_CLEANUP_LOOPEXIT]], label %[[FOR_BODY]]
+;
+entry:
+  %cmp5.not = icmp eq i64 %n, 0
+  br i1 %cmp5.not, label %for.cond.cleanup, label %for.body.preheader
+
+for.body.preheader:                               ; preds = %entry
+  br label %for.body
+
+for.cond.cleanup.loopexit:                        ; preds = %for.body
+  br label %for.cond.cleanup
+
+for.cond.cleanup:                                 ; preds = %for.cond.cleanup.loopexit, %entry
+  ret void
+
+for.body:                                         ; preds = %for.body.preheader, %for.body
+  %i.06 = phi i64 [ %inc, %for.body ], [ 0, %for.body.preheader ]
+  %mul = mul i64 %i.06, %stride
+  %arrayidx = getelementptr inbounds nuw [4 x i8], ptr %src, i64 %mul
+  %0 = load i32, ptr %arrayidx, align 4
+  %arrayidx1 = getelementptr inbounds nuw [4 x i8], ptr %dst, i64 %i.06
+  store i32 %0, ptr %arrayidx1, align 4
+  %inc = add nuw i64 %i.06, 1
+  %exitcond.not = icmp eq i64 %inc, %n
+  br i1 %exitcond.not, label %for.cond.cleanup.loopexit, label %for.body
+}
 
 define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-LABEL: define void @reversed-loop(
@@ -316,7 +489,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    [[SMIN1:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP1]], i32 -1)
 ; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[TMP0]], [[SMIN1]]
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
 ; CHECK:       [[VECTOR_SCEVCHECK]]:
 ; CHECK-NEXT:    [[TMP3:%.*]] = add i32 [[N]], -2
 ; CHECK-NEXT:    [[SMIN:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP3]], i32 -1)
@@ -352,7 +525,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
 ; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <8 x i32> poison, i32 [[VAL]], i64 0
 ; CHECK-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT3]], <8 x i32> poison, <8 x i32> zeroinitializer
 ; CHECK-NEXT:    [[REVERSE5:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLAT4]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
@@ -396,7 +569,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-VS-NEXT:    [[TMP3:%.*]] = call i32 @llvm.vscale.i32()
 ; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i32 [[TMP3]], 3
 ; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], [[TMP4]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
 ; CHECK-VS:       [[VECTOR_SCEVCHECK]]:
 ; CHECK-VS-NEXT:    [[TMP5:%.*]] = add i32 [[N]], -2
 ; CHECK-VS-NEXT:    [[SMIN:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP5]], i32 -1)
@@ -438,7 +611,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
 ; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-VS-NEXT:    [[TMP21:%.*]] = call i32 @llvm.vscale.i32()
 ; CHECK-VS-NEXT:    [[TMP22:%.*]] = shl nuw i32 [[TMP21]], 3
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[VAL]], i64 0

>From 892845e36c1344f1eddc7cbeb4fb1382a6e877d6 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Tue, 22 Sep 2026 18:10:59 +0000
Subject: [PATCH 22/25] resolve review comments

---
 .../Vectorize/LoopVectorizationPlanner.h      |  2 -
 .../Transforms/Vectorize/LoopVectorize.cpp    | 40 +++++--------------
 .../AArch64/fold-epilogue-tail.ll             | 34 ----------------
 .../LoopVectorize/fold-epilogue-tail.ll       |  6 +--
 .../Transforms/Vectorize/VPlanTest.cpp        | 18 ---------
 5 files changed, 12 insertions(+), 88 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 847a9026aa307..d870f049162db 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -893,8 +893,6 @@ class LoopVectorizationPlanner {
 
   /// The interleaved access analysis.
   InterleavedAccessInfo &IAI;
-  /// The interleaved access analysis for the case of tail-folded epilogue.
-  std::unique_ptr<InterleavedAccessInfo> EpilogueTfIAI;
 
   PredicatedScalarEvolution &PSE;
 
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index e939b164e638d..ff2d68b28dbc1 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5473,17 +5473,9 @@ bool LoopVectorizationPlanner::planForEpilogueTF() {
     return false;
   LLVM_DEBUG(dbgs() << "LV: epilogue tail-folding is enabled\n");
 
-  bool UseInterleaved = TTI.enableInterleavedAccessVectorization();
-  if (EnableInterleavedMemAccesses.getNumOccurrences() > 0)
-    UseInterleaved = EnableInterleavedMemAccesses;
-
-  EpilogueTfIAI = std::make_unique<InterleavedAccessInfo>(
-      PSE, OrigLoop, DT, LI, Legal->getLAI(), Config.OptForSize);
-  if (UseInterleaved)
-    EpilogueTfIAI->analyzeInterleaving(useMaskedInterleavedAccesses(TTI));
   LoopVectorizationCostModel EpilogueTfCM(
       EpilogueTailLoweringStatus, OrigLoop, PSE, LI, Legal, TTI, TLI, CM->AC,
-      ORE, CM->GetBFI, CM->TheFunction, *EpilogueTfIAI, Config);
+      ORE, CM->GetBFI, CM->TheFunction, IAI, Config);
 
   assert(EpilogueTfCM.preferTailFoldedLoop() &&
          "Epilogue tail-folding is expected to be enabled");
@@ -5496,7 +5488,7 @@ bool LoopVectorizationPlanner::planForEpilogueTF() {
   FixedScalableVFPair MaxFactors =
       EpilogueTfCM.computeMaxVF(EpilogueVectorizationForceVF, /*UserIC*/ 1);
   if (!MaxFactors || !EpilogueTfCM.foldTailByMasking()) {
-    // Cases that should not to be vectorized or tail-folded.
+    // Cases that should not be vectorized or tail-folded.
     reportVectorizationInfo("This case of epilogue loop can't be tail-folded",
                             "InvalidTailFoldedEpilogue", ORE, OrigLoop);
     return false;
@@ -5512,20 +5504,10 @@ bool LoopVectorizationPlanner::planForEpilogueTF() {
     reportVectorizationInfo(
         "Failed to build initial tail-folded epilogue VPlan",
         "InvalidTailFoldedEpilogue", ORE, OrigLoop);
-    assert(false && "Failed to build initial tail-folded epilogue VPlan");
+    llvm_unreachable("Failed to build initial tail-folded epilogue VPlan");
     return false;
   }
 
-  if (!useMaskedInterleavedAccesses(TTI)) {
-    LLVM_DEBUG(
-        dbgs() << "LV: Invalidate all interleaved groups due to fold-tail by "
-                  "masking which requires masked-interleaved support.\n");
-    if (EpilogueTfCM.InterleaveInfo.invalidateGroups())
-      // Invalidating interleave groups also requires invalidating all decisions
-      // based on them, which includes widening decisions and uniform and scalar
-      // values.
-      EpilogueTfCM.invalidateCostModelingDecisions();
-  }
   Legal->prepareToFoldTailByMasking();
 
   // Collect the instructions (and their associated costs) that will be more
@@ -5541,6 +5523,7 @@ bool LoopVectorizationPlanner::planForEpilogueTF() {
   if (VPlans.size() == NumPlansBefore) {
     reportVectorizationInfo("Failed to build tail-folded epilogue VPlan",
                             "InvalidTailFoldedEpilogue", ORE, OrigLoop);
+    llvm_unreachable("Failed to build tail-folded epilogue VPlan");
     return false;
   }
   if (VPlans.back()->getSingleVF() != EpilogueVectorizationForceVF ||
@@ -5548,6 +5531,7 @@ bool LoopVectorizationPlanner::planForEpilogueTF() {
     reportVectorizationInfo(
         "Failed to build a valid tail-folded epilogue VPlan",
         "InvalidTailFoldedEpilogue", ORE, OrigLoop);
+    llvm_unreachable("Failed to build a valid tail-folded epilogue VPlan");
     VPlans.pop_back();
     return false;
   }
@@ -7679,14 +7663,9 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
       // instead, so the mask reflects how many elements the main vector loop
       // already processed.
       VPBuilder EntryBuilder(Plan.getVectorPreheader());
-      Type *CanIVTy = VectorLoop->getCanonicalIVType();
-      VPValue *ALMMultiplier = Plan.getConstantInt(CanIVTy, 1);
-      auto *EntryIncrement = EntryBuilder.createOverflowingOp(
-          VPInstruction::CanonicalIVIncrementForPart, {VPV, &Plan.getVF()}, {},
-          R.getDebugLoc(), "index.part.next");
       auto *EntryALM = EntryBuilder.createNaryOp(
-          VPInstruction::WideActiveLaneMask,
-          {EntryIncrement, Plan.getTripCount(), ALMMultiplier}, R.getDebugLoc(),
+          VPInstruction::ActiveLaneMask,
+          {VPV, Plan.getTripCount()}, R.getDebugLoc(),
           "active.lane.mask.entry");
       cast<VPHeaderPHIRecipe>(&R)->setStartValue(EntryALM);
       continue;
@@ -7774,8 +7753,7 @@ fixScalarResumeValuesFromBypass(BasicBlock *BypassBlock, VPlan &BestEpiPlan,
 /// count check of the main loop, as well as updating various phis. \p
 /// InstsToMove contains instructions that need to be moved to the preheader of
 /// the epilogue vector loop.
-static void connectEpilogueVectorLoop(VPlan &EpiPlan, Loop *L,
-                                      DominatorTree *DT, LoopInfo *LI,
+static void connectEpilogueVectorLoop(VPlan &EpiPlan, DominatorTree *DT,
                                       GeneratedRTChecks &Checks,
                                       VPIRBasicBlock *VecEpilogueIterCheckVPBB,
                                       ArrayRef<Instruction *> InstsToMove,
@@ -8340,7 +8318,7 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     LVP.executePlan(
         EPI.EpilogueVF, EPI.EpilogueUF, BestEpiPlan, EpilogILV, DT,
         LoopVectorizationPlanner::EpilogueVectorizationKind::Epilogue);
-    connectEpilogueVectorLoop(BestEpiPlan, L, DT, LI, Checks,
+    connectEpilogueVectorLoop(BestEpiPlan, DT, Checks,
                               EpilogILV.VecEpilogueIterationCountCheck,
                               InstsToMove, ResumeValues, IsTailFolded);
     ++LoopsEpilogueVectorized;
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 2679cfcc7aedf..695957f5925bc 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -17,11 +17,6 @@
 ; RUN: -pass-remarks-analysis=loop-vectorize < %s 2>&1 | FileCheck %s \
 ; RUN: --check-prefix=CHECK-INVALID-COSTS
 
-; RUN: opt -S -p loop-vectorize -debug-only=loop-vectorize,vectorutils \
-; RUN: -epilogue-tail-folding-policy=prefer-fold-tail --disable-output \
-; RUN: -force-vector-width=16 -epilogue-vectorization-force-VF=8 < %s 2>&1 \
-; RUN: | FileCheck %s --check-prefix=CHECK-INVALIDATE-INTERLEAVE
-
 target triple = "aarch64-linux-gnu"
 
 define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
@@ -686,32 +681,3 @@ for.end:
   ret void
 }
 declare void @foo(ptr)
-
-
-define i64 @test_no_masked_interleave_support(i64 %y, i32 %n) {
-; CHECK-INVALIDATE-INTERLEAVE-LABEL: Checking a loop in 'test_no_masked_interleave_support'
-; CHECK-INVALIDATE-INTERLEAVE: LV: epilogue tail-folding is enabled
-; CHECK-INVALIDATE-INTERLEAVE: LV: Analyzing interleaved accesses...
-; CHECK-INVALIDATE-INTERLEAVE: LV: Invalidate all interleaved groups due to fold-tail by masking which requires masked-interleaved support
-entry:
-  br label %for.body
-
-for.body:
-  %i = phi i32 [ 0, %entry ], [ %inc, %cond.end ]
-  %cmp = icmp eq i64 %y, 0
-  br i1 %cmp, label %cond.end, label %cond.false
-
-cond.false:
-  %div = xor i64 3, %y
-  br label %cond.end
-
-cond.end:
-  %cond = phi i64 [ %div, %cond.false ], [ 77, %for.body ]
-  %inc = add nuw nsw i32 %i, 1
-  %exitcond = icmp eq i32 %inc, %n
-  br i1 %exitcond, label %for.cond.cleanup, label %for.body
-
-for.cond.cleanup:
-  ret i64 %cond
-}
-
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index 70878c60811ca..41741addb6334 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -14,7 +14,7 @@
 
 ; RUN: %{cmd} -force-vector-width=8 -epilogue-vectorization-force-VF=8 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALID-VFs
 
-; RUN: %{cmd} -force-vector-width=8 -epilogue-vectorization-force-VF=16 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALID-BIGER-EPILOGUE
+; RUN: %{cmd} -force-vector-width=8 -epilogue-vectorization-force-VF=16 < %s 2>&1 | FileCheck %s --check-prefix=CHECK-INVALID-LARGER-EPILOGUE
 
 ; RUN: %{cmd} -force-vector-width=16 -epilogue-vectorization-force-VF=8 -enable-early-exit-vectorization-with-side-effects \
 ; RUN: < %s 2>&1 | FileCheck %s --check-prefix=CHECK-DISABLED-EARLY-EXIT
@@ -47,8 +47,8 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i8 %val) {
 ; CHECK-ALIAS-MASK-LABEL: Checking a loop in 'test_epilogue_tf'
 ; CHECK-ALIAS-MASK: remark: <unknown>:0:0: Epilogue tail-folding is not supported with alias masking
 ;
-; CHECK-INVALID-BIGER-EPILOGUE-LABEL: Checking a loop in 'test_epilogue_tf'
-; CHECK-INVALID-BIGER-EPILOGUE: remark: <unknown>:0:0: For now, epilogue tail-folding can't be applied when VF of the main loop <= VF of the epilogue
+; CHECK-INVALID-LARGER-EPILOGUE-LABEL: Checking a loop in 'test_epilogue_tf'
+; CHECK-INVALID-LARGER-EPILOGUE: remark: <unknown>:0:0: For now, epilogue tail-folding can't be applied when VF of the main loop <= VF of the epilogue
 
 entry:
   br label %for.body
diff --git a/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp b/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
index 5a62cd122265d..ffc3a0d303474 100644
--- a/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
+++ b/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
@@ -1760,24 +1760,6 @@ TEST_F(VPRecipeTest, CastVPReductionEVLRecipeToVPUser) {
   VPReductionEVLRecipe EVLRecipe(Recipe, *EVL, CondOp);
   checkVPRecipeCastImpl<VPReductionEVLRecipe, VPUser>(&EVLRecipe);
 }
-
-TEST_F(VPRecipeTest, CastVPWidenCanonicalIVRecipeToVPUser) {
-  VPlan &Plan = getPlan();
-  VPBasicBlock *Preheader = Plan.getEntry();
-  VPBasicBlock *Header = Plan.createVPBasicBlock("header");
-  VPBasicBlock *Latch = Plan.createVPBasicBlock("latch");
-  VPRegionBlock *Region = Plan.createLoopRegion(Type::getInt32Ty(C), DebugLoc(),
-                                                "loop", Header, Latch);
-  VPBlockUtils::connectBlocks(Header, Latch);
-  VPBlockUtils::connectBlocks(Preheader, Region);
-  VPBlockUtils::connectBlocks(Region, Plan.getScalarHeader());
-
-  VPRegionValue *CanIV = Region->getCanonicalIV();
-  VPWidenCanonicalIVRecipe Recipe(CanIV);
-
-  EXPECT_EQ(CanIV, Recipe.getCanonicalIV());
-  checkVPRecipeCastImpl<VPWidenCanonicalIVRecipe, VPUser>(&Recipe);
-}
 } // namespace
 
 struct VPDoubleValueDef : public VPRecipeBase {

>From 5f0d22f86f9db33b56777aa9e43acea731a618c2 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Tue, 22 Sep 2026 18:16:06 +0000
Subject: [PATCH 23/25] format

---
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 16 ++++++++++------
 1 file changed, 10 insertions(+), 6 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ff2d68b28dbc1..51a9744a2d27d 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -1114,6 +1114,12 @@ class LoopVectorizationCostModel {
     return EpilogueLoweringStatus == CM_EpilogueAllowed;
   }
 
+  /// Returns true if tail-folding is preferred over an epilogue.
+  bool preferTailFoldedLoop() const {
+    return EpilogueLoweringStatus == CM_EpilogueNotNeededFoldTail ||
+           EpilogueLoweringStatus == CM_EpilogueNotAllowedFoldTail;
+  }
+
   /// Returns the TailFoldingStyle that is best for the current loop.
   TailFoldingStyle getTailFoldingStyle() const {
     return ChosenTailFoldingStyle;
@@ -5465,9 +5471,8 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
 }
 
 bool LoopVectorizationPlanner::planForEpilogueTF() {
-  EpilogueLowering EpilogueTailLoweringStatus =
-      getEpilogueTailLowering(*CM, OrigLoop, ORE, *Legal, Config.getHints(),
-                              &TTI);
+  EpilogueLowering EpilogueTailLoweringStatus = getEpilogueTailLowering(
+      *CM, OrigLoop, ORE, *Legal, Config.getHints(), &TTI);
   if (EpilogueTailLoweringStatus !=
       EpilogueLowering::CM_EpilogueNotNeededFoldTail)
     return false;
@@ -7664,9 +7669,8 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
       // already processed.
       VPBuilder EntryBuilder(Plan.getVectorPreheader());
       auto *EntryALM = EntryBuilder.createNaryOp(
-          VPInstruction::ActiveLaneMask,
-          {VPV, Plan.getTripCount()}, R.getDebugLoc(),
-          "active.lane.mask.entry");
+          VPInstruction::ActiveLaneMask, {VPV, Plan.getTripCount()},
+          R.getDebugLoc(), "active.lane.mask.entry");
       cast<VPHeaderPHIRecipe>(&R)->setStartValue(EntryALM);
       continue;
     } else {

>From cf75b376c9bb5496706ef726efd8dc4ea1d5a277 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Sun, 27 Sep 2026 21:36:59 +0000
Subject: [PATCH 24/25] update after rebase

---
 .../Transforms/Vectorize/LoopVectorize.cpp    |  2 +-
 .../AArch64/fold-epilogue-tail.ll             | 20 ++---
 .../LoopVectorize/AArch64/sve-widen-phi.ll    | 82 +++++++++----------
 3 files changed, 52 insertions(+), 52 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 51a9744a2d27d..f2001be64d922 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -6554,7 +6554,7 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(
   RUN_VPLAN_PASS(VPlanTransforms::removeDeadRecipes, *VPlan0);
   if (IsInnerLoop) {
     RUN_VPLAN_PASS(VPlanTransforms::recordExecutionFrequencies, *VPlan0);
-    assert(verifyExecutionFrequenciesMatchBFI(*VPlan0, OrigLoop, LI, *CM) &&
+    assert(verifyExecutionFrequenciesMatchBFI(*VPlan0, OrigLoop, LI, EnabledCM) &&
            "execution frequencies do not match the loop's block frequencies");
   }
 
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index 695957f5925bc..ff0595fc3a73c 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -49,7 +49,7 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
 ; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <8 x i32> poison, i32 [[VAL]], i64 0
 ; CHECK-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT2]], <8 x i32> poison, <8 x i32> zeroinitializer
@@ -111,7 +111,7 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
 ; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
 ; CHECK-VS-NEXT:    [[TMP7:%.*]] = call i64 @llvm.vscale.i64()
 ; CHECK-VS-NEXT:    [[TMP8:%.*]] = shl nuw i64 [[TMP7]], 3
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
@@ -185,7 +185,7 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
 ; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
@@ -251,7 +251,7 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
 ; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
 ; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i64 @llvm.vscale.i64()
 ; CHECK-VS-NEXT:    [[TMP12:%.*]] = shl nuw i64 [[TMP11]], 3
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
@@ -477,7 +477,7 @@ for.body:                                         ; preds = %for.body.preheader,
 define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-LABEL: define void @reversed-loop(
 ; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:  [[ITER_CHECK:.*:]]
 ; CHECK-NEXT:    [[ST:%.*]] = sub i32 [[N]], 1
 ; CHECK-NEXT:    [[TMP0:%.*]] = add i32 [[N]], -1
 ; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[N]], -2
@@ -491,10 +491,10 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    [[TMP4:%.*]] = sub i32 [[TMP3]], [[SMIN]]
 ; CHECK-NEXT:    [[TMP5:%.*]] = sub i32 [[ST]], [[TMP4]]
 ; CHECK-NEXT:    [[TMP6:%.*]] = icmp sgt i32 [[TMP5]], [[ST]]
-; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK2:%.*]] = icmp ult i32 [[TMP2]], 32
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[TMP7:%.*]] = and i32 [[TMP2]], 31
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP2]], [[TMP7]]
@@ -555,7 +555,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ;
 ; CHECK-VS-LABEL: define void @reversed-loop(
 ; CHECK-VS-SAME: ptr [[A:%.*]], i32 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0]] {
-; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*:]]
 ; CHECK-VS-NEXT:    [[ST:%.*]] = sub i32 [[N]], 1
 ; CHECK-VS-NEXT:    [[TMP0:%.*]] = add i32 [[N]], -1
 ; CHECK-VS-NEXT:    [[TMP1:%.*]] = add i32 [[N]], -2
@@ -571,11 +571,11 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-VS-NEXT:    [[TMP6:%.*]] = sub i32 [[TMP5]], [[SMIN]]
 ; CHECK-VS-NEXT:    [[TMP7:%.*]] = sub i32 [[ST]], [[TMP6]]
 ; CHECK-VS-NEXT:    [[TMP8:%.*]] = icmp sgt i32 [[TMP7]], [[ST]]
-; CHECK-VS-NEXT:    br i1 [[TMP8]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS-NEXT:    br i1 [[TMP8]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
 ; CHECK-VS-NEXT:    [[TMP9:%.*]] = shl nuw i32 [[TMP3]], 5
 ; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK2:%.*]] = icmp ult i32 [[TMP2]], [[TMP9]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
 ; CHECK-VS-NEXT:    [[TMP10:%.*]] = shl nuw i32 [[TMP3]], 4
 ; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i32 [[TMP2]], [[TMP9]]
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
index ab86f398ebd2e..c76cb4ebce44c 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
@@ -110,65 +110,65 @@ define void @widen_ptr_phi_unrolled(ptr noalias nocapture %a, ptr noalias nocapt
 ; CHECK-EPI-TF:       vector.body:
 ; CHECK-EPI-TF-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
 ; CHECK-EPI-TF-NEXT:    [[TMP6:%.*]] = shl i64 [[INDEX]], 3
-; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = mul i64 [[TMP3]], 8
-; CHECK-EPI-TF-NEXT:    [[TMP9:%.*]] = add i64 [[TMP6]], [[TMP8]]
+; CHECK-EPI-TF-NEXT:    [[TMP7:%.*]] = mul i64 [[TMP3]], 8
+; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = add i64 [[TMP6]], [[TMP7]]
 ; CHECK-EPI-TF-NEXT:    [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP6]]
-; CHECK-EPI-TF-NEXT:    [[NEXT_GEP2:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP9]]
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP2:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP8]]
 ; CHECK-EPI-TF-NEXT:    [[WIDE_VEC:%.*]] = load <vscale x 8 x i32>, ptr [[NEXT_GEP]], align 4
 ; CHECK-EPI-TF-NEXT:    [[STRIDED_VEC:%.*]] = call { <vscale x 4 x i32>, <vscale x 4 x i32> } @llvm.vector.deinterleave2.nxv8i32(<vscale x 8 x i32> [[WIDE_VEC]])
-; CHECK-EPI-TF-NEXT:    [[TMP10:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 0
-; CHECK-EPI-TF-NEXT:    [[TMP11:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP9:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 0
+; CHECK-EPI-TF-NEXT:    [[TMP10:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 1
 ; CHECK-EPI-TF-NEXT:    [[WIDE_VEC3:%.*]] = load <vscale x 8 x i32>, ptr [[NEXT_GEP2]], align 4
 ; CHECK-EPI-TF-NEXT:    [[STRIDED_VEC4:%.*]] = call { <vscale x 4 x i32>, <vscale x 4 x i32> } @llvm.vector.deinterleave2.nxv8i32(<vscale x 8 x i32> [[WIDE_VEC3]])
-; CHECK-EPI-TF-NEXT:    [[TMP12:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC4]], 0
-; CHECK-EPI-TF-NEXT:    [[TMP13:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC4]], 1
-; CHECK-EPI-TF-NEXT:    [[TMP14:%.*]] = add nsw <vscale x 4 x i32> [[TMP10]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP15:%.*]] = add nsw <vscale x 4 x i32> [[TMP12]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[A:%.*]], i64 [[INDEX]]
-; CHECK-EPI-TF-NEXT:    [[TMP17:%.*]] = getelementptr inbounds i32, ptr [[TMP16]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    [[TMP11:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC4]], 0
+; CHECK-EPI-TF-NEXT:    [[TMP12:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC4]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP13:%.*]] = add nsw <vscale x 4 x i32> [[TMP9]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP14:%.*]] = add nsw <vscale x 4 x i32> [[TMP11]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[A:%.*]], i64 [[INDEX]]
+; CHECK-EPI-TF-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[TMP15]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP13]], ptr [[TMP15]], align 4
 ; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP14]], ptr [[TMP16]], align 4
-; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP15]], ptr [[TMP17]], align 4
-; CHECK-EPI-TF-NEXT:    [[TMP18:%.*]] = add nsw <vscale x 4 x i32> [[TMP11]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP19:%.*]] = add nsw <vscale x 4 x i32> [[TMP13]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[B:%.*]], i64 [[INDEX]]
-; CHECK-EPI-TF-NEXT:    [[TMP21:%.*]] = getelementptr inbounds i32, ptr [[TMP20]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    [[TMP17:%.*]] = add nsw <vscale x 4 x i32> [[TMP10]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP18:%.*]] = add nsw <vscale x 4 x i32> [[TMP12]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP19:%.*]] = getelementptr inbounds i32, ptr [[B:%.*]], i64 [[INDEX]]
+; CHECK-EPI-TF-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[TMP19]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP17]], ptr [[TMP19]], align 4
 ; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP18]], ptr [[TMP20]], align 4
-; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP19]], ptr [[TMP21]], align 4
 ; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; CHECK-EPI-TF-NEXT:    [[TMP22:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-EPI-TF-NEXT:    br i1 [[TMP22]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-EPI-TF-NEXT:    [[TMP21:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP21]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
 ; CHECK-EPI-TF:       middle.block:
 ; CHECK-EPI-TF-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-EPI-TF-NEXT:    br i1 [[CMP_N]], label [[FOR_EXIT:%.*]], label [[VEC_EPILOG_ITER_CHECK:%.*]]
 ; CHECK-EPI-TF:       vec.epilog.iter.check:
 ; CHECK-EPI-TF-NEXT:    br i1 false, label [[VEC_EPILOG_SCALAR_PH:%.*]], label [[VEC_EPILOG_PH]]
 ; CHECK-EPI-TF:       vec.epilog.ph:
-; CHECK-EPI-TF-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[ITER_CHECK:%.*]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-EPI-TF-NEXT:    [[TMP23:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-EPI-TF-NEXT:    [[TMP24:%.*]] = shl nuw i64 [[TMP23]], 1
+; CHECK-EPI-TF-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, [[ITER_CHECK:%.*]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP22:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-EPI-TF-NEXT:    [[TMP23:%.*]] = shl nuw i64 [[TMP22]], 1
 ; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-EPI-TF-NEXT:    br label [[VEC_EPILOG_VECTOR_BODY:%.*]]
 ; CHECK-EPI-TF:       vec.epilog.vector.body:
 ; CHECK-EPI-TF-NEXT:    [[INDEX5:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 2 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], [[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-EPI-TF-NEXT:    [[TMP25:%.*]] = shl i64 [[INDEX5]], 3
-; CHECK-EPI-TF-NEXT:    [[NEXT_GEP6:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP25]]
+; CHECK-EPI-TF-NEXT:    [[TMP24:%.*]] = shl i64 [[INDEX5]], 3
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP6:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP24]]
 ; CHECK-EPI-TF-NEXT:    [[INTERLEAVED_MASK:%.*]] = call <vscale x 4 x i1> @llvm.vector.interleave2.nxv4i1(<vscale x 2 x i1> [[ACTIVE_LANE_MASK]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
 ; CHECK-EPI-TF-NEXT:    [[WIDE_MASKED_VEC:%.*]] = call <vscale x 4 x i32> @llvm.masked.load.nxv4i32.p0(ptr align 4 [[NEXT_GEP6]], <vscale x 4 x i1> [[INTERLEAVED_MASK]], <vscale x 4 x i32> poison)
 ; CHECK-EPI-TF-NEXT:    [[STRIDED_VEC7:%.*]] = call { <vscale x 2 x i32>, <vscale x 2 x i32> } @llvm.vector.deinterleave2.nxv4i32(<vscale x 4 x i32> [[WIDE_MASKED_VEC]])
-; CHECK-EPI-TF-NEXT:    [[TMP26:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32> } [[STRIDED_VEC7]], 0
-; CHECK-EPI-TF-NEXT:    [[TMP27:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32> } [[STRIDED_VEC7]], 1
-; CHECK-EPI-TF-NEXT:    [[TMP28:%.*]] = add nsw <vscale x 2 x i32> [[TMP26]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP29:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX5]]
-; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP28]], ptr align 4 [[TMP29]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-EPI-TF-NEXT:    [[TMP30:%.*]] = add nsw <vscale x 2 x i32> [[TMP27]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP31:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX5]]
-; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP30]], ptr align 4 [[TMP31]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT8]] = add i64 [[INDEX5]], [[TMP24]]
+; CHECK-EPI-TF-NEXT:    [[TMP25:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32> } [[STRIDED_VEC7]], 0
+; CHECK-EPI-TF-NEXT:    [[TMP26:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32> } [[STRIDED_VEC7]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP27:%.*]] = add nsw <vscale x 2 x i32> [[TMP25]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP28:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX5]]
+; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP27]], ptr align 4 [[TMP28]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-EPI-TF-NEXT:    [[TMP29:%.*]] = add nsw <vscale x 2 x i32> [[TMP26]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP30:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX5]]
+; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP29]], ptr align 4 [[TMP30]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT8]] = add i64 [[INDEX5]], [[TMP23]]
 ; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[INDEX_NEXT8]], i64 [[N]])
-; CHECK-EPI-TF-NEXT:    [[TMP32:%.*]] = extractelement <vscale x 2 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-EPI-TF-NEXT:    [[TMP33:%.*]] = xor i1 [[TMP32]], true
-; CHECK-EPI-TF-NEXT:    br i1 [[TMP33]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-EPI-TF-NEXT:    [[TMP31:%.*]] = extractelement <vscale x 2 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-EPI-TF-NEXT:    [[TMP32:%.*]] = xor i1 [[TMP31]], true
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP32]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
 ; CHECK-EPI-TF:       vec.epilog.middle.block:
 ; CHECK-EPI-TF-NEXT:    br label [[FOR_EXIT]]
 ; CHECK-EPI-TF:       vec.epilog.scalar.ph:
@@ -177,13 +177,13 @@ define void @widen_ptr_phi_unrolled(ptr noalias nocapture %a, ptr noalias nocapt
 ; CHECK-EPI-TF-NEXT:    [[PTR_014:%.*]] = phi ptr [ [[INCDEC_PTR1:%.*]], [[FOR_BODY]] ], [ [[C]], [[VEC_EPILOG_SCALAR_PH]] ]
 ; CHECK-EPI-TF-NEXT:    [[I_013:%.*]] = phi i64 [ [[INC:%.*]], [[FOR_BODY]] ], [ 0, [[VEC_EPILOG_SCALAR_PH]] ]
 ; CHECK-EPI-TF-NEXT:    [[INCDEC_PTR:%.*]] = getelementptr inbounds i32, ptr [[PTR_014]], i64 1
-; CHECK-EPI-TF-NEXT:    [[TMP34:%.*]] = load i32, ptr [[PTR_014]], align 4
+; CHECK-EPI-TF-NEXT:    [[TMP33:%.*]] = load i32, ptr [[PTR_014]], align 4
 ; CHECK-EPI-TF-NEXT:    [[INCDEC_PTR1]] = getelementptr inbounds i32, ptr [[PTR_014]], i64 2
-; CHECK-EPI-TF-NEXT:    [[TMP35:%.*]] = load i32, ptr [[INCDEC_PTR]], align 4
-; CHECK-EPI-TF-NEXT:    [[ADD:%.*]] = add nsw i32 [[TMP34]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP34:%.*]] = load i32, ptr [[INCDEC_PTR]], align 4
+; CHECK-EPI-TF-NEXT:    [[ADD:%.*]] = add nsw i32 [[TMP33]], 1
 ; CHECK-EPI-TF-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[I_013]]
 ; CHECK-EPI-TF-NEXT:    store i32 [[ADD]], ptr [[ARRAYIDX]], align 4
-; CHECK-EPI-TF-NEXT:    [[ADD2:%.*]] = add nsw i32 [[TMP35]], 1
+; CHECK-EPI-TF-NEXT:    [[ADD2:%.*]] = add nsw i32 [[TMP34]], 1
 ; CHECK-EPI-TF-NEXT:    [[ARRAYIDX3:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[I_013]]
 ; CHECK-EPI-TF-NEXT:    store i32 [[ADD2]], ptr [[ARRAYIDX3]], align 4
 ; CHECK-EPI-TF-NEXT:    [[INC]] = add nuw nsw i64 [[I_013]], 1
@@ -322,7 +322,7 @@ define void @widen_2ptrs_phi_unrolled(ptr noalias nocapture %dst, ptr noalias no
 ; CHECK-EPI-TF:       vec.epilog.iter.check:
 ; CHECK-EPI-TF-NEXT:    br i1 false, label [[VEC_EPILOG_SCALAR_PH:%.*]], label [[VEC_EPILOG_PH]]
 ; CHECK-EPI-TF:       vec.epilog.ph:
-; CHECK-EPI-TF-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[ITER_CHECK:%.*]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-EPI-TF-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, [[ITER_CHECK:%.*]] ]
 ; CHECK-EPI-TF-NEXT:    [[TMP13:%.*]] = call i64 @llvm.vscale.i64()
 ; CHECK-EPI-TF-NEXT:    [[TMP14:%.*]] = shl nuw i64 [[TMP13]], 1
 ; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])

>From d98dab0311a3db70c2b71bf1f5359933e631eba2 Mon Sep 17 00:00:00 2001
From: Hassnaa Hamdi <hassnaa.hamdi at arm.com>
Date: Mon, 28 Sep 2026 10:23:23 +0000
Subject: [PATCH 25/25] Use a false min-iters check instead of rewiring
 iter.check and remove not needed code from connectEpiloguePlan

---
 .../Vectorize/LoopVectorizationPlanner.h      |   8 +-
 .../Transforms/Vectorize/LoopVectorize.cpp    |  73 +---
 .../AArch64/fold-epilogue-tail.ll             | 378 +++++++++---------
 .../LoopVectorize/AArch64/sve-widen-phi.ll    | 200 +++++----
 4 files changed, 305 insertions(+), 354 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index d870f049162db..99a5864eb59fe 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -1023,9 +1023,13 @@ class LoopVectorizationPlanner {
   void emitInvalidCostRemarks(OptimizationRemarkEmitter *ORE);
 
   /// Create a check to \p Plan to see if the vector loop should be executed
-  /// based on its trip count.
+  /// based on its trip count. When \p TailFoldedEpilogue is true, the
+  /// tail-folded vector epilogue executes all iterations not executed by
+  /// the main vector loop, so the check never needs to branch to the scalar
+  /// loop.
   void addMinimumIterationCheck(VPlan &Plan, ElementCount VF, unsigned UF,
-                                ElementCount MinProfitableTripCount) const;
+                                ElementCount MinProfitableTripCount,
+                                bool TailFoldedEpilogue) const;
 
   /// Attach the runtime checks of \p RTChecks to \p Plan.
   void attachRuntimeChecks(VPlan &Plan, GeneratedRTChecks &RTChecks,
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index f2001be64d922..c9f331fcc6e1f 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -6554,8 +6554,9 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1(
   RUN_VPLAN_PASS(VPlanTransforms::removeDeadRecipes, *VPlan0);
   if (IsInnerLoop) {
     RUN_VPLAN_PASS(VPlanTransforms::recordExecutionFrequencies, *VPlan0);
-    assert(verifyExecutionFrequenciesMatchBFI(*VPlan0, OrigLoop, LI, EnabledCM) &&
-           "execution frequencies do not match the loop's block frequencies");
+    assert(
+        verifyExecutionFrequenciesMatchBFI(*VPlan0, OrigLoop, LI, EnabledCM) &&
+        "execution frequencies do not match the loop's block frequencies");
   }
 
   // Create recipes for header phis. For outer loops, reductions, recurrences
@@ -7150,14 +7151,18 @@ void LoopVectorizationPlanner::attachRuntimeChecks(
 
 void LoopVectorizationPlanner::addMinimumIterationCheck(
     VPlan &Plan, ElementCount VF, unsigned UF,
-    ElementCount MinProfitableTripCount) const {
+    ElementCount MinProfitableTripCount, bool TailFoldedEpilogue) const {
   const uint32_t *BranchWeights =
       hasBranchWeightMD(*OrigLoop->getLoopLatch()->getTerminator())
           ? &MinItersBypassWeights[0]
           : nullptr;
+  // If either the main vector loop or the epilogue vector loop is tail-folded,
+  // it can handle any trip count, so the min iter check never needs
+  // to branch to the scalar loop.
   RUN_VPLAN_PASS(VPlanTransforms::addMinimumIterationCheck, Plan, VF, UF,
                  MinProfitableTripCount, Plan.requiresScalarEpilogue(),
-                 Plan.hasTailFolded(), OrigLoop, BranchWeights,
+                 Plan.hasTailFolded() || TailFoldedEpilogue, OrigLoop,
+                 BranchWeights,
                  OrigLoop->getLoopPredecessor()->getTerminator()->getDebugLoc(),
                  PSE, Plan.getEntry());
 }
@@ -7739,6 +7744,8 @@ fixScalarResumeValuesFromBypass(BasicBlock *BypassBlock, VPlan &BestEpiPlan,
     for (auto [ResumeV, HeaderPhi] :
          zip(ResumeValues, BestEpiPlan.getScalarHeader()->phis())) {
       auto *HeaderPhiR = cast<VPIRPhi>(&HeaderPhi);
+      // With a tail-folded epilogue, the epilogue's middle block never branches
+      // to the scalar preheader, so some incoming values are just 0, not phis.
       if (!isa<PHINode>(HeaderPhiR->getIRPhi().getIncomingValueForBlock(PH)))
         continue;
       auto *EpiResumePhi =
@@ -7758,11 +7765,9 @@ fixScalarResumeValuesFromBypass(BasicBlock *BypassBlock, VPlan &BestEpiPlan,
 /// InstsToMove contains instructions that need to be moved to the preheader of
 /// the epilogue vector loop.
 static void connectEpilogueVectorLoop(VPlan &EpiPlan, DominatorTree *DT,
-                                      GeneratedRTChecks &Checks,
                                       VPIRBasicBlock *VecEpilogueIterCheckVPBB,
                                       ArrayRef<Instruction *> InstsToMove,
-                                      ArrayRef<VPInstruction *> ResumeValues,
-                                      bool IsEpilogueTfEnabled) {
+                                      ArrayRef<VPInstruction *> ResumeValues) {
   ArrayRef<VPBlockBase *> Preds = VecEpilogueIterCheckVPBB->getPredecessors();
   BasicBlock *MainLoopIterationCountCheck =
       cast<VPIRBasicBlock>(Preds.front())->getIRBasicBlock();
@@ -7780,36 +7785,6 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, DominatorTree *DT,
                     {DominatorTree::Insert, MainLoopIterationCountCheck,
                      VecEpiloguePreHeader}});
 
-  BasicBlock *ScalarPH =
-      cast<VPIRBasicBlock>(EpiPlan.getScalarPreheader())->getIRBasicBlock();
-  BasicBlock *SCEVCheckBlock = Checks.getSCEVChecks().second;
-  BasicBlock *MemCheckBlock = Checks.getMemRuntimeChecks().second;
-  // The epilogue plan's entry wraps the main loop's iteration count check
-  // (iter.check), which was redirected to the scalar preheader when executing
-  // the epilogue plan.
-  BasicBlock *EpilogueIterationCountCheck =
-      cast<VPIRBasicBlock>(EpiPlan.getEntry())->getIRBasicBlock();
-  bool JumpToScalarPH =
-      (!IsEpilogueTfEnabled || SCEVCheckBlock || MemCheckBlock);
-  // With tail-folding, the epilogue vector loop itself safely handles any
-  // trip count (including one smaller than the epilogue VF), so there's no
-  // need for a scalar remainder and we can jump straight to the epilogue
-  // preheader. The exception is when a SCEV or memory runtime check is
-  // present: those can fail at runtime regardless of tail-folding, so the
-  // scalar loop must still be kept as a fallback.
-  if (!JumpToScalarPH) {
-    assert(is_contained(successors(EpilogueIterationCountCheck), ScalarPH) &&
-           "expected iter.check to branch to the scalar preheader");
-    ScalarPH->removePredecessor(EpilogueIterationCountCheck,
-                                /*KeepOneInputPHIs=*/true);
-    EpilogueIterationCountCheck->getTerminator()->replaceSuccessorWith(
-        ScalarPH, VecEpiloguePreHeader);
-    DTU.applyUpdates(
-        {{DominatorTree::Delete, EpilogueIterationCountCheck, ScalarPH},
-         {DominatorTree::Insert, EpilogueIterationCountCheck,
-          VecEpiloguePreHeader}});
-  }
-
   // The vec.epilog.iter.check block may contain Phi nodes from inductions
   // or reductions which merge control-flow from the latch block and the
   // middle block. Update the incoming values here and move the Phi into the
@@ -7822,18 +7797,6 @@ static void connectEpilogueVectorLoop(VPlan &EpiPlan, DominatorTree *DT,
     Phi->replaceIncomingBlockWith(
         VecEpilogueIterationCountCheck->getSinglePredecessor(),
         VecEpilogueIterationCountCheck);
-    // When the epilogue is tail-folded, EpilogueIterationCountCheck
-    // (iter.check) is redirected to branch straight into the vector epilogue
-    // preheader (see the IsEpilogueTfEnabled redirect above), so it is now a
-    // genuine predecessor. Like MainLoopIterationCountCheck, it bypasses the
-    // main vector loop, so re-use the incoming value from that edge.
-    // TODO: revisit for reduction phis, whose resume value on this bypass
-    // edge may need dedicated handling rather than reusing the value already
-    // present here.
-    if (!JumpToScalarPH)
-      Phi->addIncoming(
-          Phi->getIncomingValueForBlock(MainLoopIterationCountCheck),
-          EpilogueIterationCountCheck);
   }
 
   auto IP = VecEpiloguePreHeader->getFirstNonPHIIt();
@@ -8290,7 +8253,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     // Add minimum iteration check for the epilogue plan, followed by runtime
     // checks for the main plan.
     LVP.addMinimumIterationCheck(BestMainPlan, EPI.EpilogueVF, EPI.EpilogueUF,
-                                 ElementCount::getFixed(0));
+                                 ElementCount::getFixed(0),
+                                 BestEpiPlan.hasTailFolded());
     LVP.attachRuntimeChecks(BestMainPlan, Checks, HasBranchWeights);
     RUN_VPLAN_PASS(
         VPlanTransforms::addIterationCountCheckBlock, BestMainPlan,
@@ -8317,20 +8281,19 @@ bool LoopVectorizePass::processLoop(Loop *L) {
         BestMainPlan, BestEpiPlan, L, ExpandedSCEVs, EPI, LVP, Config,
         *PSE.getSE(), ResumeValues);
     RUN_VPLAN_PASS(VPlanTransforms::simplifyLiveInsWithSCEV, BestEpiPlan, PSE);
-    // Save the status of epilogue tail-folding:
-    const bool IsTailFolded = BestEpiPlan.hasTailFolded();
     LVP.executePlan(
         EPI.EpilogueVF, EPI.EpilogueUF, BestEpiPlan, EpilogILV, DT,
         LoopVectorizationPlanner::EpilogueVectorizationKind::Epilogue);
-    connectEpilogueVectorLoop(BestEpiPlan, DT, Checks,
+    connectEpilogueVectorLoop(BestEpiPlan, DT,
                               EpilogILV.VecEpilogueIterationCountCheck,
-                              InstsToMove, ResumeValues, IsTailFolded);
+                              InstsToMove, ResumeValues);
     ++LoopsEpilogueVectorized;
   } else {
     InnerLoopVectorizer LB(L, PSE, LI, DT, TTI, AC, VF.Width, IC, Checks,
                            BestPlan);
     LVP.addMinimumIterationCheck(BestPlan, VF.Width, IC,
-                                 VF.MinProfitableTripCount);
+                                 VF.MinProfitableTripCount,
+                                 /*TailFoldedEpilogue=*/false);
     LVP.attachRuntimeChecks(BestPlan, Checks, HasBranchWeights);
 
     if (!IsInnerLoop)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
index ff0595fc3a73c..49613085e94ce 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/fold-epilogue-tail.ll
@@ -22,12 +22,11 @@ target triple = "aarch64-linux-gnu"
 define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-LABEL: define void @test_epilogue_tf(
 ; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-NEXT:  [[ITER_CHECK:.*:]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
@@ -47,20 +46,20 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <8 x i32> poison, i32 [[VAL]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT2]], <8 x i32> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT2:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT1]], <8 x i32> poison, <8 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT5:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[INDEX3:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX4]]
-; CHECK-NEXT:    call void @llvm.masked.store.v8i32.p0(<8 x i32> [[BROADCAST_SPLAT3]], ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-NEXT:    [[INDEX_NEXT5]] = add i64 [[INDEX4]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT5]], i64 [[N]])
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX3]]
+; CHECK-NEXT:    call void @llvm.masked.store.v8i32.p0(<8 x i32> [[BROADCAST_SPLAT2]], ptr align 4 [[TMP4]], <8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-NEXT:    [[INDEX_NEXT4]] = add i64 [[INDEX3]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT4]], i64 [[N]])
 ; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP6:%.*]] = xor i1 [[TMP5]], true
 ; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
@@ -80,54 +79,52 @@ define void @test_epilogue_tf(ptr %A, i64 %n, i32 %val) {
 ;
 ; CHECK-VS-LABEL: define void @test_epilogue_tf(
 ; CHECK-VS-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i32 [[VAL:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*:]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
 ; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 5
 ; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 5
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
 ; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[VAL]], i64 0
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
 ; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VECTOR_BODY]]:
 ; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds i32, ptr [[TMP4]], i64 [[TMP3]]
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds i32, ptr [[TMP3]], i64 [[TMP2]]
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP3]], align 4
 ; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP4]], align 4
-; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[BROADCAST_SPLAT]], ptr [[TMP5]], align 4
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP5]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-VS:       [[MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP8:%.*]] = shl nuw i64 [[TMP7]], 3
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = shl nuw i64 [[TMP6]], 3
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[VAL]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT2]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT2:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT1]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
 ; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT5:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[INDEX3:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP9:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX4]]
-; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[BROADCAST_SPLAT3]], ptr align 4 [[TMP9]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-VS-NEXT:    [[INDEX_NEXT5]] = add i64 [[INDEX4]], [[TMP8]]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT5]], i64 [[N]])
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = xor i1 [[TMP10]], true
-; CHECK-VS-NEXT:    br i1 [[TMP11]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX3]]
+; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[BROADCAST_SPLAT2]], ptr align 4 [[TMP8]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT4]] = add i64 [[INDEX3]], [[TMP7]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT4]], i64 [[N]])
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = xor i1 [[TMP9]], true
+; CHECK-VS-NEXT:    br i1 [[TMP10]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    br label %[[EXIT]]
 ; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
@@ -160,12 +157,11 @@ exit:
 define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-LABEL: define i32 @live-out(
 ; CHECK-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT:  [[ITER_CHECK:.*]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-NEXT:  [[ITER_CHECK:.*:]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
@@ -183,18 +179,18 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
-; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[INDEX1:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT2:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX2]]
+; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX1]]
 ; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 4 [[TMP5]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
-; CHECK-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-NEXT:    [[INDEX_NEXT2]] = add i64 [[INDEX1]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT2]], i64 [[N]])
 ; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP7:%.*]] = xor i1 [[TMP6]], true
 ; CHECK-NEXT:    br i1 [[TMP7]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
@@ -219,58 +215,56 @@ define i32 @live-out(ptr %src, i64 %n) {
 ;
 ; CHECK-VS-LABEL: define i32 @live-out(
 ; CHECK-VS-SAME: ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-VS-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-VS-NEXT:  [[ITER_CHECK:.*:]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
 ; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 5
 ; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
-; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 5
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH]], label %[[VECTOR_PH:.*]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
 ; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
 ; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VECTOR_BODY]]:
 ; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP4]], i64 [[TMP3]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP5]], align 4
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP3]], i64 [[TMP2]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP4]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP5]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-VS:       [[MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vscale.i32()
-; CHECK-VS-NEXT:    [[TMP8:%.*]] = mul nuw i32 [[TMP7]], 16
-; CHECK-VS-NEXT:    [[TMP9:%.*]] = sub i32 [[TMP8]], 1
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = extractelement <vscale x 16 x i32> [[WIDE_LOAD]], i32 [[TMP9]]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = mul nuw i32 [[TMP6]], 16
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = sub i32 [[TMP7]], 1
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = extractelement <vscale x 16 x i32> [[WIDE_LOAD]], i32 [[TMP8]]
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[FOR_END:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_ITER_CHECK]]:
-; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VEC_EPILOG_PH]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
-; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = shl nuw i64 [[TMP11]], 3
+; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = shl nuw i64 [[TMP10]], 3
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[INDEX1:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT2:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX2]]
-; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 4 [[TMP13]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
-; CHECK-VS-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], [[TMP12]]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = xor i1 [[TMP14]], true
-; CHECK-VS-NEXT:    br i1 [[TMP15]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = getelementptr inbounds nuw i32, ptr [[SRC]], i64 [[INDEX1]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 4 [[TMP12]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[INDEX_NEXT2]] = add i64 [[INDEX1]], [[TMP11]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT2]], i64 [[N]])
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = xor i1 [[TMP13]], true
+; CHECK-VS-NEXT:    br i1 [[TMP14]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
-; CHECK-VS-NEXT:    [[TMP16:%.*]] = xor <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
-; CHECK-VS-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP16]], i1 false)
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = xor <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], splat (i1 true)
+; CHECK-VS-NEXT:    [[FIRST_INACTIVE_LANE:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP15]], i1 false)
 ; CHECK-VS-NEXT:    [[LAST_ACTIVE_LANE:%.*]] = sub i64 [[FIRST_INACTIVE_LANE]], 1
-; CHECK-VS-NEXT:    [[TMP17:%.*]] = extractelement <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = extractelement <vscale x 8 x i32> [[WIDE_MASKED_LOAD]], i64 [[LAST_ACTIVE_LANE]]
 ; CHECK-VS-NEXT:    br label %[[FOR_END]]
 ; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
 ; CHECK-VS-NEXT:    br label %[[LOOP:.*]]
@@ -282,7 +276,7 @@ define i32 @live-out(ptr %src, i64 %n) {
 ; CHECK-VS-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
 ; CHECK-VS-NEXT:    br i1 [[EC]], label %[[FOR_END]], label %[[LOOP]]
 ; CHECK-VS:       [[FOR_END]]:
-; CHECK-VS-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[LOAD]], %[[LOOP]] ], [ [[TMP10]], %[[MIDDLE_BLOCK]] ], [ [[TMP17]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-VS-NEXT:    [[LOAD_LCSSA:%.*]] = phi i32 [ [[LOAD]], %[[LOOP]] ], [ [[TMP9]], %[[MIDDLE_BLOCK]] ], [ [[TMP16]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
 ; CHECK-VS-NEXT:    ret i32 [[LOAD_LCSSA]]
 ;
 entry:
@@ -307,14 +301,13 @@ define void @stride_copy(ptr noalias nofree noundef writeonly captures(none) %ds
 ; CHECK-NEXT:    [[CMP5_NOT:%.*]] = icmp eq i64 [[N]], 0
 ; CHECK-NEXT:    br i1 [[CMP5_NOT]], label %[[FOR_COND_CLEANUP:.*]], label %[[ITER_CHECK:.*]]
 ; CHECK:       [[ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
 ; CHECK:       [[VECTOR_SCEVCHECK]]:
 ; CHECK-NEXT:    [[IDENT_CHECK:%.*]] = icmp ne i64 [[STRIDE]], 1
 ; CHECK-NEXT:    br i1 [[IDENT_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], 32
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[TMP0:%.*]] = and i64 [[N]], 31
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[TMP0]]
@@ -324,11 +317,11 @@ define void @stride_copy(ptr noalias nofree noundef writeonly captures(none) %ds
 ; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[INDEX]]
 ; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP1]], i64 16
 ; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <16 x i32>, ptr [[TMP1]], align 4
-; CHECK-NEXT:    [[WIDE_LOAD2:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <16 x i32>, ptr [[TMP2]], align 4
 ; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[INDEX]]
 ; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP3]], i64 16
 ; CHECK-NEXT:    store <16 x i32> [[WIDE_LOAD]], ptr [[TMP3]], align 4
-; CHECK-NEXT:    store <16 x i32> [[WIDE_LOAD2]], ptr [[TMP4]], align 4
+; CHECK-NEXT:    store <16 x i32> [[WIDE_LOAD1]], ptr [[TMP4]], align 4
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 32
 ; CHECK-NEXT:    [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP5]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
@@ -342,14 +335,14 @@ define void @stride_copy(ptr noalias nofree noundef writeonly captures(none) %ds
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX3:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[INDEX3]]
+; CHECK-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[INDEX2]]
 ; CHECK-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 4 [[TMP6]], <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i32> poison)
-; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[INDEX3]]
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[INDEX2]]
 ; CHECK-NEXT:    call void @llvm.masked.store.v8i32.p0(<8 x i32> [[WIDE_MASKED_LOAD]], ptr align 4 [[TMP7]], <8 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-NEXT:    [[INDEX_NEXT4]] = add i64 [[INDEX3]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT4]], i64 [[N]])
+; CHECK-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
 ; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP9:%.*]] = xor i1 [[TMP8]], true
 ; CHECK-NEXT:    br i1 [[TMP9]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
@@ -378,35 +371,33 @@ define void @stride_copy(ptr noalias nofree noundef writeonly captures(none) %ds
 ; CHECK-VS-NEXT:    [[CMP5_NOT:%.*]] = icmp eq i64 [[N]], 0
 ; CHECK-VS-NEXT:    br i1 [[CMP5_NOT]], label %[[FOR_COND_CLEANUP:.*]], label %[[ITER_CHECK:.*]]
 ; CHECK-VS:       [[ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
 ; CHECK-VS:       [[VECTOR_SCEVCHECK]]:
 ; CHECK-VS-NEXT:    [[IDENT_CHECK:%.*]] = icmp ne i64 [[STRIDE]], 1
 ; CHECK-VS-NEXT:    br i1 [[IDENT_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 5
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK-VS-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 4
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; CHECK-VS-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 4
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
 ; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
 ; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VECTOR_BODY]]:
 ; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP4]], i64 [[TMP3]]
-; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP4]], align 4
-; CHECK-VS-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 16 x i32>, ptr [[TMP5]], align 4
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP6]], i64 [[TMP3]]
-; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[WIDE_LOAD]], ptr [[TMP6]], align 4
-; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[WIDE_LOAD2]], ptr [[TMP7]], align 4
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; CHECK-VS-NEXT:    [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP3]], i64 [[TMP2]]
+; CHECK-VS-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 16 x i32>, ptr [[TMP3]], align 4
+; CHECK-VS-NEXT:    [[WIDE_LOAD1:%.*]] = load <vscale x 16 x i32>, ptr [[TMP4]], align 4
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = getelementptr inbounds nuw i32, ptr [[TMP5]], i64 [[TMP2]]
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[WIDE_LOAD]], ptr [[TMP5]], align 4
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[WIDE_LOAD1]], ptr [[TMP6]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-VS:       [[MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[FOR_COND_CLEANUP_LOOPEXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
@@ -414,22 +405,22 @@ define void @stride_copy(ptr noalias nofree noundef writeonly captures(none) %ds
 ; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP9:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = shl nuw i64 [[TMP9]], 3
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = shl nuw i64 [[TMP8]], 3
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX3:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[INDEX2:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[INDEX3]]
-; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 4 [[TMP11]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[INDEX3]]
-; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[WIDE_MASKED_LOAD]], ptr align 4 [[TMP12]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-VS-NEXT:    [[INDEX_NEXT4]] = add i64 [[INDEX3]], [[TMP10]]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT4]], i64 [[N]])
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = xor i1 [[TMP13]], true
-; CHECK-VS-NEXT:    br i1 [[TMP14]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[INDEX2]]
+; CHECK-VS-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i32> @llvm.masked.load.nxv8i32.p0(ptr align 4 [[TMP10]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]], <vscale x 8 x i32> poison)
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[INDEX2]]
+; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[WIDE_MASKED_LOAD]], ptr align 4 [[TMP11]], <vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT3]] = add i64 [[INDEX2]], [[TMP9]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 [[INDEX_NEXT3]], i64 [[N]])
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = xor i1 [[TMP12]], true
+; CHECK-VS-NEXT:    br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    br label %[[FOR_COND_CLEANUP_LOOPEXIT]]
 ; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
@@ -442,9 +433,9 @@ define void @stride_copy(ptr noalias nofree noundef writeonly captures(none) %ds
 ; CHECK-VS-NEXT:    [[I_06:%.*]] = phi i64 [ [[INC:%.*]], %[[FOR_BODY]] ], [ 0, %[[VEC_EPILOG_SCALAR_PH]] ]
 ; CHECK-VS-NEXT:    [[MUL:%.*]] = mul i64 [[I_06]], [[STRIDE]]
 ; CHECK-VS-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[MUL]]
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = load i32, ptr [[ARRAYIDX]], align 4
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = load i32, ptr [[ARRAYIDX]], align 4
 ; CHECK-VS-NEXT:    [[ARRAYIDX1:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[I_06]]
-; CHECK-VS-NEXT:    store i32 [[TMP15]], ptr [[ARRAYIDX1]], align 4
+; CHECK-VS-NEXT:    store i32 [[TMP14]], ptr [[ARRAYIDX1]], align 4
 ; CHECK-VS-NEXT:    [[INC]] = add nuw i64 [[I_06]], 1
 ; CHECK-VS-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INC]], [[N]]
 ; CHECK-VS-NEXT:    br i1 [[EXITCOND_NOT]], label %[[FOR_COND_CLEANUP_LOOPEXIT]], label %[[FOR_BODY]]
@@ -483,8 +474,7 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[N]], -2
 ; CHECK-NEXT:    [[SMIN1:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP1]], i32 -1)
 ; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[TMP0]], [[SMIN1]]
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 8
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
 ; CHECK:       [[VECTOR_SCEVCHECK]]:
 ; CHECK-NEXT:    [[TMP3:%.*]] = add i32 [[N]], -2
 ; CHECK-NEXT:    [[SMIN:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP3]], i32 -1)
@@ -493,8 +483,8 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    [[TMP6:%.*]] = icmp sgt i32 [[TMP5]], [[ST]]
 ; CHECK-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-NEXT:    [[MIN_ITERS_CHECK2:%.*]] = icmp ult i32 [[TMP2]], 32
-; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], 32
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[TMP7:%.*]] = and i32 [[TMP2]], 31
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP2]], [[TMP7]]
@@ -521,21 +511,21 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK:       [[VEC_EPILOG_PH]]:
 ; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <8 x i32> poison, i32 [[VAL]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT3]], <8 x i32> poison, <8 x i32> zeroinitializer
-; CHECK-NEXT:    [[REVERSE5:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLAT4]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLATINSERT2]], <8 x i32> poison, <8 x i32> zeroinitializer
+; CHECK-NEXT:    [[REVERSE4:%.*]] = shufflevector <8 x i32> [[BROADCAST_SPLAT3]], <8 x i32> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[VEC_EPILOG_RESUME_VAL]], i32 [[TMP2]])
 ; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-NEXT:    [[INDEX6:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[INDEX5:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT7:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP14:%.*]] = sub i32 [[ST]], [[INDEX6]]
+; CHECK-NEXT:    [[TMP14:%.*]] = sub i32 [[ST]], [[INDEX5]]
 ; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP14]]
 ; CHECK-NEXT:    [[TMP16:%.*]] = getelementptr i32, ptr [[TMP15]], i64 -7
-; CHECK-NEXT:    [[REVERSE7:%.*]] = shufflevector <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
-; CHECK-NEXT:    call void @llvm.masked.store.v8i32.p0(<8 x i32> [[REVERSE5]], ptr align 4 [[TMP16]], <8 x i1> [[REVERSE7]])
-; CHECK-NEXT:    [[INDEX_NEXT8]] = add i32 [[INDEX6]], 8
-; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[INDEX_NEXT8]], i32 [[TMP2]])
+; CHECK-NEXT:    [[REVERSE6:%.*]] = shufflevector <8 x i1> [[ACTIVE_LANE_MASK]], <8 x i1> poison, <8 x i32> <i32 7, i32 6, i32 5, i32 4, i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    call void @llvm.masked.store.v8i32.p0(<8 x i32> [[REVERSE4]], ptr align 4 [[TMP16]], <8 x i1> [[REVERSE6]])
+; CHECK-NEXT:    [[INDEX_NEXT7]] = add i32 [[INDEX5]], 8
+; CHECK-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i32(i32 [[INDEX_NEXT7]], i32 [[TMP2]])
 ; CHECK-NEXT:    [[TMP17:%.*]] = extractelement <8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
 ; CHECK-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
 ; CHECK-NEXT:    br i1 [[TMP18]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
@@ -561,45 +551,43 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-VS-NEXT:    [[TMP1:%.*]] = add i32 [[N]], -2
 ; CHECK-VS-NEXT:    [[SMIN1:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP1]], i32 -1)
 ; CHECK-VS-NEXT:    [[TMP2:%.*]] = sub i32 [[TMP0]], [[SMIN1]]
-; CHECK-VS-NEXT:    [[TMP3:%.*]] = call i32 @llvm.vscale.i32()
-; CHECK-VS-NEXT:    [[TMP4:%.*]] = shl nuw i32 [[TMP3]], 3
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], [[TMP4]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
 ; CHECK-VS:       [[VECTOR_SCEVCHECK]]:
-; CHECK-VS-NEXT:    [[TMP5:%.*]] = add i32 [[N]], -2
-; CHECK-VS-NEXT:    [[SMIN:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP5]], i32 -1)
-; CHECK-VS-NEXT:    [[TMP6:%.*]] = sub i32 [[TMP5]], [[SMIN]]
-; CHECK-VS-NEXT:    [[TMP7:%.*]] = sub i32 [[ST]], [[TMP6]]
-; CHECK-VS-NEXT:    [[TMP8:%.*]] = icmp sgt i32 [[TMP7]], [[ST]]
-; CHECK-VS-NEXT:    br i1 [[TMP8]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK-VS-NEXT:    [[TMP3:%.*]] = add i32 [[N]], -2
+; CHECK-VS-NEXT:    [[SMIN:%.*]] = call i32 @llvm.smin.i32(i32 [[TMP3]], i32 -1)
+; CHECK-VS-NEXT:    [[TMP4:%.*]] = sub i32 [[TMP3]], [[SMIN]]
+; CHECK-VS-NEXT:    [[TMP5:%.*]] = sub i32 [[ST]], [[TMP4]]
+; CHECK-VS-NEXT:    [[TMP6:%.*]] = icmp sgt i32 [[TMP5]], [[ST]]
+; CHECK-VS-NEXT:    br i1 [[TMP6]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
 ; CHECK-VS:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
-; CHECK-VS-NEXT:    [[TMP9:%.*]] = shl nuw i32 [[TMP3]], 5
-; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK2:%.*]] = icmp ult i32 [[TMP2]], [[TMP9]]
-; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK-VS-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP8:%.*]] = shl nuw i32 [[TMP7]], 5
+; CHECK-VS-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP2]], [[TMP8]]
+; CHECK-VS-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
 ; CHECK-VS:       [[VECTOR_PH]]:
-; CHECK-VS-NEXT:    [[TMP10:%.*]] = shl nuw i32 [[TMP3]], 4
-; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i32 [[TMP2]], [[TMP9]]
+; CHECK-VS-NEXT:    [[TMP9:%.*]] = shl nuw i32 [[TMP7]], 4
+; CHECK-VS-NEXT:    [[N_MOD_VF:%.*]] = urem i32 [[TMP2]], [[TMP8]]
 ; CHECK-VS-NEXT:    [[N_VEC:%.*]] = sub i32 [[TMP2]], [[N_MOD_VF]]
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[VAL]], i64 0
 ; CHECK-VS-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-VS-NEXT:    [[TMP11:%.*]] = sub i32 [[ST]], [[N_VEC]]
+; CHECK-VS-NEXT:    [[TMP10:%.*]] = sub i32 [[ST]], [[N_VEC]]
 ; CHECK-VS-NEXT:    [[REVERSE:%.*]] = call <vscale x 16 x i32> @llvm.vector.reverse.nxv16i32(<vscale x 16 x i32> [[BROADCAST_SPLAT]])
 ; CHECK-VS-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VECTOR_BODY]]:
 ; CHECK-VS-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP12:%.*]] = sub i32 [[ST]], [[INDEX]]
-; CHECK-VS-NEXT:    [[TMP13:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP12]]
-; CHECK-VS-NEXT:    [[TMP14:%.*]] = zext i32 [[TMP10]] to i64
-; CHECK-VS-NEXT:    [[TMP15:%.*]] = sub nuw nsw i64 [[TMP14]], 1
-; CHECK-VS-NEXT:    [[TMP16:%.*]] = sub i64 0, [[TMP15]]
-; CHECK-VS-NEXT:    [[TMP17:%.*]] = getelementptr inbounds i32, ptr [[TMP13]], i64 [[TMP16]]
-; CHECK-VS-NEXT:    [[TMP18:%.*]] = sub i64 [[TMP16]], [[TMP14]]
-; CHECK-VS-NEXT:    [[TMP19:%.*]] = getelementptr inbounds i32, ptr [[TMP13]], i64 [[TMP18]]
-; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP17]], align 4
-; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP19]], align 4
-; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], [[TMP9]]
-; CHECK-VS-NEXT:    [[TMP20:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-VS-NEXT:    br i1 [[TMP20]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP11:%.*]] = sub i32 [[ST]], [[INDEX]]
+; CHECK-VS-NEXT:    [[TMP12:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP11]]
+; CHECK-VS-NEXT:    [[TMP13:%.*]] = zext i32 [[TMP9]] to i64
+; CHECK-VS-NEXT:    [[TMP14:%.*]] = sub nuw nsw i64 [[TMP13]], 1
+; CHECK-VS-NEXT:    [[TMP15:%.*]] = sub i64 0, [[TMP14]]
+; CHECK-VS-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[TMP12]], i64 [[TMP15]]
+; CHECK-VS-NEXT:    [[TMP17:%.*]] = sub i64 [[TMP15]], [[TMP13]]
+; CHECK-VS-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[TMP12]], i64 [[TMP17]]
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP16]], align 4
+; CHECK-VS-NEXT:    store <vscale x 16 x i32> [[REVERSE]], ptr [[TMP18]], align 4
+; CHECK-VS-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], [[TMP8]]
+; CHECK-VS-NEXT:    [[TMP19:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-VS-NEXT:    br i1 [[TMP19]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]]
 ; CHECK-VS:       [[MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[TMP2]], [[N_VEC]]
 ; CHECK-VS-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
@@ -607,29 +595,29 @@ define void @reversed-loop(ptr %A, i32 %n, i32 %val) {
 ; CHECK-VS-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]]
 ; CHECK-VS:       [[VEC_EPILOG_PH]]:
 ; CHECK-VS-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-VS-NEXT:    [[TMP21:%.*]] = call i32 @llvm.vscale.i32()
-; CHECK-VS-NEXT:    [[TMP22:%.*]] = shl nuw i32 [[TMP21]], 3
-; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[VAL]], i64 0
-; CHECK-VS-NEXT:    [[BROADCAST_SPLAT4:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT3]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
-; CHECK-VS-NEXT:    [[REVERSE5:%.*]] = call <vscale x 8 x i32> @llvm.vector.reverse.nxv8i32(<vscale x 8 x i32> [[BROADCAST_SPLAT4]])
+; CHECK-VS-NEXT:    [[TMP20:%.*]] = call i32 @llvm.vscale.i32()
+; CHECK-VS-NEXT:    [[TMP21:%.*]] = shl nuw i32 [[TMP20]], 3
+; CHECK-VS-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <vscale x 8 x i32> poison, i32 [[VAL]], i64 0
+; CHECK-VS-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <vscale x 8 x i32> [[BROADCAST_SPLATINSERT2]], <vscale x 8 x i32> poison, <vscale x 8 x i32> zeroinitializer
+; CHECK-VS-NEXT:    [[REVERSE4:%.*]] = call <vscale x 8 x i32> @llvm.vector.reverse.nxv8i32(<vscale x 8 x i32> [[BROADCAST_SPLAT3]])
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i32(i32 [[VEC_EPILOG_RESUME_VAL]], i32 [[TMP2]])
 ; CHECK-VS-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
 ; CHECK-VS:       [[VEC_EPILOG_VECTOR_BODY]]:
-; CHECK-VS-NEXT:    [[INDEX6:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-VS-NEXT:    [[INDEX5:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT7:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 8 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], %[[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-VS-NEXT:    [[TMP23:%.*]] = sub i32 [[ST]], [[INDEX6]]
-; CHECK-VS-NEXT:    [[TMP24:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP23]]
-; CHECK-VS-NEXT:    [[TMP25:%.*]] = zext i32 [[TMP22]] to i64
-; CHECK-VS-NEXT:    [[TMP26:%.*]] = sub nuw nsw i64 [[TMP25]], 1
-; CHECK-VS-NEXT:    [[TMP27:%.*]] = sub i64 0, [[TMP26]]
-; CHECK-VS-NEXT:    [[TMP28:%.*]] = getelementptr i32, ptr [[TMP24]], i64 [[TMP27]]
-; CHECK-VS-NEXT:    [[REVERSE7:%.*]] = call <vscale x 8 x i1> @llvm.vector.reverse.nxv8i1(<vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[REVERSE5]], ptr align 4 [[TMP28]], <vscale x 8 x i1> [[REVERSE7]])
-; CHECK-VS-NEXT:    [[INDEX_NEXT8]] = add i32 [[INDEX6]], [[TMP22]]
-; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i32(i32 [[INDEX_NEXT8]], i32 [[TMP2]])
-; CHECK-VS-NEXT:    [[TMP29:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-VS-NEXT:    [[TMP30:%.*]] = xor i1 [[TMP29]], true
-; CHECK-VS-NEXT:    br i1 [[TMP30]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
+; CHECK-VS-NEXT:    [[TMP22:%.*]] = sub i32 [[ST]], [[INDEX5]]
+; CHECK-VS-NEXT:    [[TMP23:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[TMP22]]
+; CHECK-VS-NEXT:    [[TMP24:%.*]] = zext i32 [[TMP21]] to i64
+; CHECK-VS-NEXT:    [[TMP25:%.*]] = sub nuw nsw i64 [[TMP24]], 1
+; CHECK-VS-NEXT:    [[TMP26:%.*]] = sub i64 0, [[TMP25]]
+; CHECK-VS-NEXT:    [[TMP27:%.*]] = getelementptr i32, ptr [[TMP23]], i64 [[TMP26]]
+; CHECK-VS-NEXT:    [[REVERSE6:%.*]] = call <vscale x 8 x i1> @llvm.vector.reverse.nxv8i1(<vscale x 8 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-VS-NEXT:    call void @llvm.masked.store.nxv8i32.p0(<vscale x 8 x i32> [[REVERSE4]], ptr align 4 [[TMP27]], <vscale x 8 x i1> [[REVERSE6]])
+; CHECK-VS-NEXT:    [[INDEX_NEXT7]] = add i32 [[INDEX5]], [[TMP21]]
+; CHECK-VS-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i32(i32 [[INDEX_NEXT7]], i32 [[TMP2]])
+; CHECK-VS-NEXT:    [[TMP28:%.*]] = extractelement <vscale x 8 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-VS-NEXT:    [[TMP29:%.*]] = xor i1 [[TMP28]], true
+; CHECK-VS-NEXT:    br i1 [[TMP29]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]]
 ; CHECK-VS:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
 ; CHECK-VS-NEXT:    br label %[[EXIT]]
 ; CHECK-VS:       [[VEC_EPILOG_SCALAR_PH]]:
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
index c76cb4ebce44c..d2894a1c0943a 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-widen-phi.ll
@@ -92,83 +92,81 @@ define void @widen_ptr_phi_unrolled(ptr noalias nocapture %a, ptr noalias nocapt
 ;
 ; CHECK-EPI-TF-LABEL: @widen_ptr_phi_unrolled(
 ; CHECK-EPI-TF-NEXT:  iter.check:
+; CHECK-EPI-TF-NEXT:    br i1 false, label [[VEC_EPILOG_SCALAR_PH:%.*]], label [[VECTOR_MAIN_LOOP_ITER_CHECK:%.*]]
+; CHECK-EPI-TF:       vector.main.loop.iter.check:
 ; CHECK-EPI-TF-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-EPI-TF-NEXT:    [[TMP1:%.*]] = shl nuw nsw i64 [[TMP0]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
 ; CHECK-EPI-TF-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N:%.*]], [[TMP1]]
-; CHECK-EPI-TF-NEXT:    br i1 [[MIN_ITERS_CHECK]], label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_MAIN_LOOP_ITER_CHECK:%.*]]
-; CHECK-EPI-TF:       vector.main.loop.iter.check:
-; CHECK-EPI-TF-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 3
-; CHECK-EPI-TF-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
-; CHECK-EPI-TF-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label [[VEC_EPILOG_PH]], label [[VECTOR_PH:%.*]]
+; CHECK-EPI-TF-NEXT:    br i1 [[MIN_ITERS_CHECK]], label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_PH:%.*]]
 ; CHECK-EPI-TF:       vector.ph:
-; CHECK-EPI-TF-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 2
-; CHECK-EPI-TF-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; CHECK-EPI-TF-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-EPI-TF-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
 ; CHECK-EPI-TF-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
-; CHECK-EPI-TF-NEXT:    [[TMP4:%.*]] = shl i64 [[N_VEC]], 3
-; CHECK-EPI-TF-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[C:%.*]], i64 [[TMP4]]
+; CHECK-EPI-TF-NEXT:    [[TMP3:%.*]] = shl i64 [[N_VEC]], 3
+; CHECK-EPI-TF-NEXT:    [[TMP4:%.*]] = getelementptr i8, ptr [[C:%.*]], i64 [[TMP3]]
 ; CHECK-EPI-TF-NEXT:    br label [[VECTOR_BODY:%.*]]
 ; CHECK-EPI-TF:       vector.body:
 ; CHECK-EPI-TF-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-EPI-TF-NEXT:    [[TMP6:%.*]] = shl i64 [[INDEX]], 3
-; CHECK-EPI-TF-NEXT:    [[TMP7:%.*]] = mul i64 [[TMP3]], 8
-; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = add i64 [[TMP6]], [[TMP7]]
-; CHECK-EPI-TF-NEXT:    [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP6]]
-; CHECK-EPI-TF-NEXT:    [[NEXT_GEP2:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP8]]
+; CHECK-EPI-TF-NEXT:    [[TMP5:%.*]] = shl i64 [[INDEX]], 3
+; CHECK-EPI-TF-NEXT:    [[TMP6:%.*]] = mul i64 [[TMP2]], 8
+; CHECK-EPI-TF-NEXT:    [[TMP7:%.*]] = add i64 [[TMP5]], [[TMP6]]
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP5]]
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP1:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP7]]
 ; CHECK-EPI-TF-NEXT:    [[WIDE_VEC:%.*]] = load <vscale x 8 x i32>, ptr [[NEXT_GEP]], align 4
 ; CHECK-EPI-TF-NEXT:    [[STRIDED_VEC:%.*]] = call { <vscale x 4 x i32>, <vscale x 4 x i32> } @llvm.vector.deinterleave2.nxv8i32(<vscale x 8 x i32> [[WIDE_VEC]])
-; CHECK-EPI-TF-NEXT:    [[TMP9:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 0
-; CHECK-EPI-TF-NEXT:    [[TMP10:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 1
-; CHECK-EPI-TF-NEXT:    [[WIDE_VEC3:%.*]] = load <vscale x 8 x i32>, ptr [[NEXT_GEP2]], align 4
-; CHECK-EPI-TF-NEXT:    [[STRIDED_VEC4:%.*]] = call { <vscale x 4 x i32>, <vscale x 4 x i32> } @llvm.vector.deinterleave2.nxv8i32(<vscale x 8 x i32> [[WIDE_VEC3]])
-; CHECK-EPI-TF-NEXT:    [[TMP11:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC4]], 0
-; CHECK-EPI-TF-NEXT:    [[TMP12:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC4]], 1
-; CHECK-EPI-TF-NEXT:    [[TMP13:%.*]] = add nsw <vscale x 4 x i32> [[TMP9]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP14:%.*]] = add nsw <vscale x 4 x i32> [[TMP11]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[A:%.*]], i64 [[INDEX]]
-; CHECK-EPI-TF-NEXT:    [[TMP16:%.*]] = getelementptr inbounds i32, ptr [[TMP15]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 0
+; CHECK-EPI-TF-NEXT:    [[TMP9:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC]], 1
+; CHECK-EPI-TF-NEXT:    [[WIDE_VEC2:%.*]] = load <vscale x 8 x i32>, ptr [[NEXT_GEP1]], align 4
+; CHECK-EPI-TF-NEXT:    [[STRIDED_VEC3:%.*]] = call { <vscale x 4 x i32>, <vscale x 4 x i32> } @llvm.vector.deinterleave2.nxv8i32(<vscale x 8 x i32> [[WIDE_VEC2]])
+; CHECK-EPI-TF-NEXT:    [[TMP10:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC3]], 0
+; CHECK-EPI-TF-NEXT:    [[TMP11:%.*]] = extractvalue { <vscale x 4 x i32>, <vscale x 4 x i32> } [[STRIDED_VEC3]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP12:%.*]] = add nsw <vscale x 4 x i32> [[TMP8]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP13:%.*]] = add nsw <vscale x 4 x i32> [[TMP10]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP14:%.*]] = getelementptr inbounds i32, ptr [[A:%.*]], i64 [[INDEX]]
+; CHECK-EPI-TF-NEXT:    [[TMP15:%.*]] = getelementptr inbounds i32, ptr [[TMP14]], i64 [[TMP2]]
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP12]], ptr [[TMP14]], align 4
 ; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP13]], ptr [[TMP15]], align 4
-; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP14]], ptr [[TMP16]], align 4
-; CHECK-EPI-TF-NEXT:    [[TMP17:%.*]] = add nsw <vscale x 4 x i32> [[TMP10]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP18:%.*]] = add nsw <vscale x 4 x i32> [[TMP12]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP19:%.*]] = getelementptr inbounds i32, ptr [[B:%.*]], i64 [[INDEX]]
-; CHECK-EPI-TF-NEXT:    [[TMP20:%.*]] = getelementptr inbounds i32, ptr [[TMP19]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    [[TMP16:%.*]] = add nsw <vscale x 4 x i32> [[TMP9]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP17:%.*]] = add nsw <vscale x 4 x i32> [[TMP11]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP18:%.*]] = getelementptr inbounds i32, ptr [[B:%.*]], i64 [[INDEX]]
+; CHECK-EPI-TF-NEXT:    [[TMP19:%.*]] = getelementptr inbounds i32, ptr [[TMP18]], i64 [[TMP2]]
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP16]], ptr [[TMP18]], align 4
 ; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP17]], ptr [[TMP19]], align 4
-; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP18]], ptr [[TMP20]], align 4
-; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; CHECK-EPI-TF-NEXT:    [[TMP21:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-EPI-TF-NEXT:    br i1 [[TMP21]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; CHECK-EPI-TF-NEXT:    [[TMP20:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP20]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
 ; CHECK-EPI-TF:       middle.block:
 ; CHECK-EPI-TF-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-EPI-TF-NEXT:    br i1 [[CMP_N]], label [[FOR_EXIT:%.*]], label [[VEC_EPILOG_ITER_CHECK:%.*]]
 ; CHECK-EPI-TF:       vec.epilog.iter.check:
-; CHECK-EPI-TF-NEXT:    br i1 false, label [[VEC_EPILOG_SCALAR_PH:%.*]], label [[VEC_EPILOG_PH]]
+; CHECK-EPI-TF-NEXT:    br i1 false, label [[VEC_EPILOG_SCALAR_PH]], label [[VEC_EPILOG_PH]]
 ; CHECK-EPI-TF:       vec.epilog.ph:
-; CHECK-EPI-TF-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, [[ITER_CHECK:%.*]] ]
-; CHECK-EPI-TF-NEXT:    [[TMP22:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-EPI-TF-NEXT:    [[TMP23:%.*]] = shl nuw i64 [[TMP22]], 1
+; CHECK-EPI-TF-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP21:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-EPI-TF-NEXT:    [[TMP22:%.*]] = shl nuw i64 [[TMP21]], 1
 ; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-EPI-TF-NEXT:    br label [[VEC_EPILOG_VECTOR_BODY:%.*]]
 ; CHECK-EPI-TF:       vec.epilog.vector.body:
-; CHECK-EPI-TF-NEXT:    [[INDEX5:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT7:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 2 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], [[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-EPI-TF-NEXT:    [[TMP24:%.*]] = shl i64 [[INDEX5]], 3
-; CHECK-EPI-TF-NEXT:    [[NEXT_GEP6:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP24]]
+; CHECK-EPI-TF-NEXT:    [[TMP23:%.*]] = shl i64 [[INDEX4]], 3
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP5:%.*]] = getelementptr i8, ptr [[C]], i64 [[TMP23]]
 ; CHECK-EPI-TF-NEXT:    [[INTERLEAVED_MASK:%.*]] = call <vscale x 4 x i1> @llvm.vector.interleave2.nxv4i1(<vscale x 2 x i1> [[ACTIVE_LANE_MASK]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-EPI-TF-NEXT:    [[WIDE_MASKED_VEC:%.*]] = call <vscale x 4 x i32> @llvm.masked.load.nxv4i32.p0(ptr align 4 [[NEXT_GEP6]], <vscale x 4 x i1> [[INTERLEAVED_MASK]], <vscale x 4 x i32> poison)
-; CHECK-EPI-TF-NEXT:    [[STRIDED_VEC7:%.*]] = call { <vscale x 2 x i32>, <vscale x 2 x i32> } @llvm.vector.deinterleave2.nxv4i32(<vscale x 4 x i32> [[WIDE_MASKED_VEC]])
-; CHECK-EPI-TF-NEXT:    [[TMP25:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32> } [[STRIDED_VEC7]], 0
-; CHECK-EPI-TF-NEXT:    [[TMP26:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32> } [[STRIDED_VEC7]], 1
-; CHECK-EPI-TF-NEXT:    [[TMP27:%.*]] = add nsw <vscale x 2 x i32> [[TMP25]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP28:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX5]]
-; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP27]], ptr align 4 [[TMP28]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-EPI-TF-NEXT:    [[TMP29:%.*]] = add nsw <vscale x 2 x i32> [[TMP26]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP30:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX5]]
-; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP29]], ptr align 4 [[TMP30]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT8]] = add i64 [[INDEX5]], [[TMP23]]
-; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[INDEX_NEXT8]], i64 [[N]])
-; CHECK-EPI-TF-NEXT:    [[TMP31:%.*]] = extractelement <vscale x 2 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-EPI-TF-NEXT:    [[TMP32:%.*]] = xor i1 [[TMP31]], true
-; CHECK-EPI-TF-NEXT:    br i1 [[TMP32]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-EPI-TF-NEXT:    [[WIDE_MASKED_VEC:%.*]] = call <vscale x 4 x i32> @llvm.masked.load.nxv4i32.p0(ptr align 4 [[NEXT_GEP5]], <vscale x 4 x i1> [[INTERLEAVED_MASK]], <vscale x 4 x i32> poison)
+; CHECK-EPI-TF-NEXT:    [[STRIDED_VEC6:%.*]] = call { <vscale x 2 x i32>, <vscale x 2 x i32> } @llvm.vector.deinterleave2.nxv4i32(<vscale x 4 x i32> [[WIDE_MASKED_VEC]])
+; CHECK-EPI-TF-NEXT:    [[TMP24:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32> } [[STRIDED_VEC6]], 0
+; CHECK-EPI-TF-NEXT:    [[TMP25:%.*]] = extractvalue { <vscale x 2 x i32>, <vscale x 2 x i32> } [[STRIDED_VEC6]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP26:%.*]] = add nsw <vscale x 2 x i32> [[TMP24]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP27:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX4]]
+; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP26]], ptr align 4 [[TMP27]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-EPI-TF-NEXT:    [[TMP28:%.*]] = add nsw <vscale x 2 x i32> [[TMP25]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP29:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX4]]
+; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP28]], ptr align 4 [[TMP29]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT7]] = add i64 [[INDEX4]], [[TMP22]]
+; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[INDEX_NEXT7]], i64 [[N]])
+; CHECK-EPI-TF-NEXT:    [[TMP30:%.*]] = extractelement <vscale x 2 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-EPI-TF-NEXT:    [[TMP31:%.*]] = xor i1 [[TMP30]], true
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP31]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
 ; CHECK-EPI-TF:       vec.epilog.middle.block:
 ; CHECK-EPI-TF-NEXT:    br label [[FOR_EXIT]]
 ; CHECK-EPI-TF:       vec.epilog.scalar.ph:
@@ -177,13 +175,13 @@ define void @widen_ptr_phi_unrolled(ptr noalias nocapture %a, ptr noalias nocapt
 ; CHECK-EPI-TF-NEXT:    [[PTR_014:%.*]] = phi ptr [ [[INCDEC_PTR1:%.*]], [[FOR_BODY]] ], [ [[C]], [[VEC_EPILOG_SCALAR_PH]] ]
 ; CHECK-EPI-TF-NEXT:    [[I_013:%.*]] = phi i64 [ [[INC:%.*]], [[FOR_BODY]] ], [ 0, [[VEC_EPILOG_SCALAR_PH]] ]
 ; CHECK-EPI-TF-NEXT:    [[INCDEC_PTR:%.*]] = getelementptr inbounds i32, ptr [[PTR_014]], i64 1
-; CHECK-EPI-TF-NEXT:    [[TMP33:%.*]] = load i32, ptr [[PTR_014]], align 4
+; CHECK-EPI-TF-NEXT:    [[TMP32:%.*]] = load i32, ptr [[PTR_014]], align 4
 ; CHECK-EPI-TF-NEXT:    [[INCDEC_PTR1]] = getelementptr inbounds i32, ptr [[PTR_014]], i64 2
-; CHECK-EPI-TF-NEXT:    [[TMP34:%.*]] = load i32, ptr [[INCDEC_PTR]], align 4
-; CHECK-EPI-TF-NEXT:    [[ADD:%.*]] = add nsw i32 [[TMP33]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP33:%.*]] = load i32, ptr [[INCDEC_PTR]], align 4
+; CHECK-EPI-TF-NEXT:    [[ADD:%.*]] = add nsw i32 [[TMP32]], 1
 ; CHECK-EPI-TF-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[I_013]]
 ; CHECK-EPI-TF-NEXT:    store i32 [[ADD]], ptr [[ARRAYIDX]], align 4
-; CHECK-EPI-TF-NEXT:    [[ADD2:%.*]] = add nsw i32 [[TMP34]], 1
+; CHECK-EPI-TF-NEXT:    [[ADD2:%.*]] = add nsw i32 [[TMP33]], 1
 ; CHECK-EPI-TF-NEXT:    [[ARRAYIDX3:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[I_013]]
 ; CHECK-EPI-TF-NEXT:    store i32 [[ADD2]], ptr [[ARRAYIDX3]], align 4
 ; CHECK-EPI-TF-NEXT:    [[INC]] = add nuw nsw i64 [[I_013]], 1
@@ -284,63 +282,61 @@ define void @widen_2ptrs_phi_unrolled(ptr noalias nocapture %dst, ptr noalias no
 ;
 ; CHECK-EPI-TF-LABEL: @widen_2ptrs_phi_unrolled(
 ; CHECK-EPI-TF-NEXT:  iter.check:
+; CHECK-EPI-TF-NEXT:    br i1 false, label [[VEC_EPILOG_SCALAR_PH:%.*]], label [[VECTOR_MAIN_LOOP_ITER_CHECK:%.*]]
+; CHECK-EPI-TF:       vector.main.loop.iter.check:
 ; CHECK-EPI-TF-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-EPI-TF-NEXT:    [[TMP1:%.*]] = shl nuw nsw i64 [[TMP0]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
 ; CHECK-EPI-TF-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[N:%.*]], [[TMP1]]
-; CHECK-EPI-TF-NEXT:    br i1 [[MIN_ITERS_CHECK]], label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_MAIN_LOOP_ITER_CHECK:%.*]]
-; CHECK-EPI-TF:       vector.main.loop.iter.check:
-; CHECK-EPI-TF-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 3
-; CHECK-EPI-TF-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP2]]
-; CHECK-EPI-TF-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label [[VEC_EPILOG_PH]], label [[VECTOR_PH:%.*]]
+; CHECK-EPI-TF-NEXT:    br i1 [[MIN_ITERS_CHECK]], label [[VEC_EPILOG_PH:%.*]], label [[VECTOR_PH:%.*]]
 ; CHECK-EPI-TF:       vector.ph:
-; CHECK-EPI-TF-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 2
-; CHECK-EPI-TF-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP2]]
+; CHECK-EPI-TF-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-EPI-TF-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
 ; CHECK-EPI-TF-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
-; CHECK-EPI-TF-NEXT:    [[TMP4:%.*]] = shl i64 [[N_VEC]], 2
-; CHECK-EPI-TF-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[SRC:%.*]], i64 [[TMP4]]
-; CHECK-EPI-TF-NEXT:    [[TMP6:%.*]] = getelementptr i8, ptr [[DST:%.*]], i64 [[TMP4]]
+; CHECK-EPI-TF-NEXT:    [[TMP3:%.*]] = shl i64 [[N_VEC]], 2
+; CHECK-EPI-TF-NEXT:    [[TMP4:%.*]] = getelementptr i8, ptr [[SRC:%.*]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    [[TMP5:%.*]] = getelementptr i8, ptr [[DST:%.*]], i64 [[TMP3]]
 ; CHECK-EPI-TF-NEXT:    br label [[VECTOR_BODY:%.*]]
 ; CHECK-EPI-TF:       vector.body:
 ; CHECK-EPI-TF-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[VECTOR_BODY]] ]
-; CHECK-EPI-TF-NEXT:    [[TMP7:%.*]] = shl i64 [[INDEX]], 2
-; CHECK-EPI-TF-NEXT:    [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP7]]
-; CHECK-EPI-TF-NEXT:    [[NEXT_GEP2:%.*]] = getelementptr i8, ptr [[DST]], i64 [[TMP7]]
-; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = getelementptr i32, ptr [[NEXT_GEP]], i64 [[TMP3]]
+; CHECK-EPI-TF-NEXT:    [[TMP6:%.*]] = shl i64 [[INDEX]], 2
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP6]]
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP1:%.*]] = getelementptr i8, ptr [[DST]], i64 [[TMP6]]
+; CHECK-EPI-TF-NEXT:    [[TMP7:%.*]] = getelementptr i32, ptr [[NEXT_GEP]], i64 [[TMP2]]
 ; CHECK-EPI-TF-NEXT:    [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[NEXT_GEP]], align 4
-; CHECK-EPI-TF-NEXT:    [[WIDE_LOAD3:%.*]] = load <vscale x 4 x i32>, ptr [[TMP8]], align 4
-; CHECK-EPI-TF-NEXT:    [[TMP9:%.*]] = shl nsw <vscale x 4 x i32> [[WIDE_LOAD]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP10:%.*]] = shl nsw <vscale x 4 x i32> [[WIDE_LOAD3]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    [[TMP11:%.*]] = getelementptr i32, ptr [[NEXT_GEP2]], i64 [[TMP3]]
-; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP9]], ptr [[NEXT_GEP2]], align 4
-; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP10]], ptr [[TMP11]], align 4
-; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP2]]
-; CHECK-EPI-TF-NEXT:    [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-EPI-TF-NEXT:    br i1 [[TMP12]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-EPI-TF-NEXT:    [[WIDE_LOAD2:%.*]] = load <vscale x 4 x i32>, ptr [[TMP7]], align 4
+; CHECK-EPI-TF-NEXT:    [[TMP8:%.*]] = shl nsw <vscale x 4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP9:%.*]] = shl nsw <vscale x 4 x i32> [[WIDE_LOAD2]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    [[TMP10:%.*]] = getelementptr i32, ptr [[NEXT_GEP1]], i64 [[TMP2]]
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP8]], ptr [[NEXT_GEP1]], align 4
+; CHECK-EPI-TF-NEXT:    store <vscale x 4 x i32> [[TMP9]], ptr [[TMP10]], align 4
+; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; CHECK-EPI-TF-NEXT:    [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP11]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
 ; CHECK-EPI-TF:       middle.block:
 ; CHECK-EPI-TF-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
 ; CHECK-EPI-TF-NEXT:    br i1 [[CMP_N]], label [[FOR_COND_CLEANUP:%.*]], label [[VEC_EPILOG_ITER_CHECK:%.*]]
 ; CHECK-EPI-TF:       vec.epilog.iter.check:
-; CHECK-EPI-TF-NEXT:    br i1 false, label [[VEC_EPILOG_SCALAR_PH:%.*]], label [[VEC_EPILOG_PH]]
+; CHECK-EPI-TF-NEXT:    br i1 false, label [[VEC_EPILOG_SCALAR_PH]], label [[VEC_EPILOG_PH]]
 ; CHECK-EPI-TF:       vec.epilog.ph:
-; CHECK-EPI-TF-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ], [ 0, [[ITER_CHECK:%.*]] ]
-; CHECK-EPI-TF-NEXT:    [[TMP13:%.*]] = call i64 @llvm.vscale.i64()
-; CHECK-EPI-TF-NEXT:    [[TMP14:%.*]] = shl nuw i64 [[TMP13]], 1
+; CHECK-EPI-TF-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[VEC_EPILOG_ITER_CHECK]] ], [ 0, [[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-EPI-TF-NEXT:    [[TMP12:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-EPI-TF-NEXT:    [[TMP13:%.*]] = shl nuw i64 [[TMP12]], 1
 ; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_ENTRY:%.*]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[VEC_EPILOG_RESUME_VAL]], i64 [[N]])
 ; CHECK-EPI-TF-NEXT:    br label [[VEC_EPILOG_VECTOR_BODY:%.*]]
 ; CHECK-EPI-TF:       vec.epilog.vector.body:
-; CHECK-EPI-TF-NEXT:    [[INDEX5:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT8:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-EPI-TF-NEXT:    [[INDEX4:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], [[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT7:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
 ; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK:%.*]] = phi <vscale x 2 x i1> [ [[ACTIVE_LANE_MASK_ENTRY]], [[VEC_EPILOG_PH]] ], [ [[ACTIVE_LANE_MASK_NEXT:%.*]], [[VEC_EPILOG_VECTOR_BODY]] ]
-; CHECK-EPI-TF-NEXT:    [[TMP15:%.*]] = shl i64 [[INDEX5]], 2
-; CHECK-EPI-TF-NEXT:    [[NEXT_GEP6:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP15]]
-; CHECK-EPI-TF-NEXT:    [[NEXT_GEP7:%.*]] = getelementptr i8, ptr [[DST]], i64 [[TMP15]]
-; CHECK-EPI-TF-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 2 x i32> @llvm.masked.load.nxv2i32.p0(ptr align 4 [[NEXT_GEP6]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]], <vscale x 2 x i32> poison)
-; CHECK-EPI-TF-NEXT:    [[TMP16:%.*]] = shl nsw <vscale x 2 x i32> [[WIDE_MASKED_LOAD]], splat (i32 1)
-; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP16]], ptr align 4 [[NEXT_GEP7]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
-; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT8]] = add i64 [[INDEX5]], [[TMP14]]
-; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[INDEX_NEXT8]], i64 [[N]])
-; CHECK-EPI-TF-NEXT:    [[TMP17:%.*]] = extractelement <vscale x 2 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
-; CHECK-EPI-TF-NEXT:    [[TMP18:%.*]] = xor i1 [[TMP17]], true
-; CHECK-EPI-TF-NEXT:    br i1 [[TMP18]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-EPI-TF-NEXT:    [[TMP14:%.*]] = shl i64 [[INDEX4]], 2
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP5:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[TMP14]]
+; CHECK-EPI-TF-NEXT:    [[NEXT_GEP6:%.*]] = getelementptr i8, ptr [[DST]], i64 [[TMP14]]
+; CHECK-EPI-TF-NEXT:    [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 2 x i32> @llvm.masked.load.nxv2i32.p0(ptr align 4 [[NEXT_GEP5]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]], <vscale x 2 x i32> poison)
+; CHECK-EPI-TF-NEXT:    [[TMP15:%.*]] = shl nsw <vscale x 2 x i32> [[WIDE_MASKED_LOAD]], splat (i32 1)
+; CHECK-EPI-TF-NEXT:    call void @llvm.masked.store.nxv2i32.p0(<vscale x 2 x i32> [[TMP15]], ptr align 4 [[NEXT_GEP6]], <vscale x 2 x i1> [[ACTIVE_LANE_MASK]])
+; CHECK-EPI-TF-NEXT:    [[INDEX_NEXT7]] = add i64 [[INDEX4]], [[TMP13]]
+; CHECK-EPI-TF-NEXT:    [[ACTIVE_LANE_MASK_NEXT]] = call <vscale x 2 x i1> @llvm.get.active.lane.mask.nxv2i1.i64(i64 [[INDEX_NEXT7]], i64 [[N]])
+; CHECK-EPI-TF-NEXT:    [[TMP16:%.*]] = extractelement <vscale x 2 x i1> [[ACTIVE_LANE_MASK_NEXT]], i64 0
+; CHECK-EPI-TF-NEXT:    [[TMP17:%.*]] = xor i1 [[TMP16]], true
+; CHECK-EPI-TF-NEXT:    br i1 [[TMP17]], label [[VEC_EPILOG_MIDDLE_BLOCK:%.*]], label [[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
 ; CHECK-EPI-TF:       vec.epilog.middle.block:
 ; CHECK-EPI-TF-NEXT:    br label [[FOR_COND_CLEANUP]]
 ; CHECK-EPI-TF:       vec.epilog.scalar.ph:
@@ -349,8 +345,8 @@ define void @widen_2ptrs_phi_unrolled(ptr noalias nocapture %dst, ptr noalias no
 ; CHECK-EPI-TF-NEXT:    [[I_011:%.*]] = phi i64 [ [[INC:%.*]], [[FOR_BODY]] ], [ 0, [[VEC_EPILOG_SCALAR_PH]] ]
 ; CHECK-EPI-TF-NEXT:    [[S_010:%.*]] = phi ptr [ [[INCDEC_PTR1:%.*]], [[FOR_BODY]] ], [ [[SRC]], [[VEC_EPILOG_SCALAR_PH]] ]
 ; CHECK-EPI-TF-NEXT:    [[D_09:%.*]] = phi ptr [ [[INCDEC_PTR:%.*]], [[FOR_BODY]] ], [ [[DST]], [[VEC_EPILOG_SCALAR_PH]] ]
-; CHECK-EPI-TF-NEXT:    [[TMP19:%.*]] = load i32, ptr [[S_010]], align 4
-; CHECK-EPI-TF-NEXT:    [[MUL:%.*]] = shl nsw i32 [[TMP19]], 1
+; CHECK-EPI-TF-NEXT:    [[TMP18:%.*]] = load i32, ptr [[S_010]], align 4
+; CHECK-EPI-TF-NEXT:    [[MUL:%.*]] = shl nsw i32 [[TMP18]], 1
 ; CHECK-EPI-TF-NEXT:    store i32 [[MUL]], ptr [[D_09]], align 4
 ; CHECK-EPI-TF-NEXT:    [[INCDEC_PTR]] = getelementptr inbounds i32, ptr [[D_09]], i64 1
 ; CHECK-EPI-TF-NEXT:    [[INCDEC_PTR1]] = getelementptr inbounds i32, ptr [[S_010]], i64 1



More information about the llvm-commits mailing list