[llvm] [LV] Allow setting scalable VFs for `-epilogue-vectorization-force-VF` (PR #205081)

Benjamin Maxwell via llvm-commits llvm-commits at lists.llvm.org
Tue Jul 7 01:48:45 PDT 2026


https://github.com/MacDue updated https://github.com/llvm/llvm-project/pull/205081

>From 29f3644d4ba97378794c0e390ecb53bfe2f18b8d Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 22 Jun 2026 11:15:10 +0000
Subject: [PATCH 1/5] [LV] Allow setting scalable VFs for
 `-epilogue-vectorization-force-VF`

Follow up to #204953. This allows replacing a unit test with an IR test.
---
 .../Transforms/Vectorize/LoopVectorize.cpp    | 27 ++++--
 ...interleave-to-widen-memory-epilogue-vec.ll | 26 -----
 ...form-narrow-interleave-vscale-x-UF-step.ll | 96 +++++++++++++++++++
 .../Transforms/Vectorize/VPlanTest.cpp        | 32 -------
 4 files changed, 113 insertions(+), 68 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 6bb68d1f7bb19..ea143b9799aea 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -178,8 +178,9 @@ static cl::opt<bool> EnableEpilogueVectorization(
     "enable-epilogue-vectorization", cl::init(true), cl::Hidden,
     cl::desc("Enable vectorization of epilogue loops."));
 
-static cl::opt<unsigned> EpilogueVectorizationForceVF(
-    "epilogue-vectorization-force-VF", cl::init(1), cl::Hidden,
+static cl::opt<ElementCount> EpilogueVectorizationForceVF(
+    "epilogue-vectorization-force-VF", cl::init(ElementCount::getFixed(1)),
+    cl::Hidden,
     cl::desc("When epilogue vectorization is enabled, and a value greater than "
              "1 is specified, forces the given VF for all applicable epilogue "
              "loops."));
@@ -417,6 +418,13 @@ static cl::opt<bool> EnableEarlyExitVectorizationWithSideEffects(
     cl::desc("Enable vectorization of early exit loops with uncountable exits "
              "and side effects"));
 
+// Returns true if the epilogue VF has been set to a non-zero value other than
+// VF=1 (scalar).
+static bool hasForcedEpilogueVF() {
+  return EpilogueVectorizationForceVF.isNonZero() &&
+         EpilogueVectorizationForceVF != ElementCount::getFixed(1);
+}
+
 // Likelyhood of bypassing the vectorized loop because there are zero trips left
 // after prolog. See `emitIterationCountCheck`.
 static constexpr uint32_t MinItersBypassWeights[] = {1, 127};
@@ -3476,8 +3484,9 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     return nullptr;
   }
 
-  if (EpilogueVectorizationForceVF > 1) {
-    if (EpilogueVectorizationForceVF >=
+  if (hasForcedEpilogueVF()) {
+    if (estimateElementCount(EpilogueVectorizationForceVF,
+                             Config.getVScaleForTuning()) >=
         IC * estimateElementCount(MainLoopVF, Config.getVScaleForTuning())) {
       // Note that the main loop leaves IC * MainLoopVF iterations iff a scalar
       // epilogue is required, but then the epilogue loop also requires a scalar
@@ -3488,7 +3497,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     }
 
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization factor is forced.\n");
-    ElementCount ForcedEC = ElementCount::getFixed(EpilogueVectorizationForceVF);
+    ElementCount ForcedEC = EpilogueVectorizationForceVF;
     if (hasPlanWithVF(ForcedEC)) {
       std::unique_ptr<VPlan> Clone(getPlanFor(ForcedEC).duplicate());
       Clone->setVF(ForcedEC);
@@ -5547,8 +5556,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
       // Collect the instructions (and their associated costs) that will be more
       // profitable to scalarize.
       CM.collectNonVectorizedAndSetWideningDecisions(UserVF);
-      ElementCount EpilogueUserVF =
-          ElementCount::getFixed(EpilogueVectorizationForceVF);
+      ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
       if (EpilogueUserVF.isVector() &&
           ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
         CM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
@@ -5801,10 +5809,9 @@ LoopVectorizationPlanner::computeBestVF() {
     return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
   }
 
-  if (hasPlanWithVF(UserVF) && EpilogueVectorizationForceVF > 1) {
+  if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF()) {
     assert(VPlans.size() == 2 && "Must have exactly 2 VPlans built");
-    assert(VPlans[0]->getSingleVF() ==
-               ElementCount::getFixed(EpilogueVectorizationForceVF) &&
+    assert(VPlans[0]->getSingleVF() == EpilogueVectorizationForceVF &&
            "expected first plan to be for the forced epilogue VF");
     assert(VPlans[1]->getSingleVF() == UserVF &&
            "expected second plan to be for the forced UserVF");
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-epilogue-vec.ll b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-epilogue-vec.ll
index f0b026e7f43a8..bea5fb0ec61d9 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-epilogue-vec.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-epilogue-vec.ll
@@ -79,29 +79,3 @@ loop:
 exit:
   ret i32 0
 }
-
-; Test that vectorization does not crash when narrowInterleaveGroups
-; materializes VFxUF and the canonical IV increment step is UF*vscale.
-%pair = type { i64, i64 }
-define void @test(ptr noalias %A, i64 %v, i64 %n) #0 {
-; CHECK-LABEL: define void @test(
-; CHECK:       vector.body:
-;
-entry:
-  br label %for.body
-
-for.body:
-  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
-  %idx.0 = getelementptr inbounds %pair, ptr %A, i64 %iv, i32 0
-  %idx.1 = getelementptr inbounds %pair, ptr %A, i64 %iv, i32 1
-  store i64 %v, ptr %idx.0
-  store i64 %v, ptr %idx.1
-  %iv.next = add nuw nsw i64 %iv, 1
-  %exitcond = icmp eq i64 %iv.next, %n
-  br i1 %exitcond, label %exit, label %for.body
-
-exit:
-  ret void
-}
-
-attributes #0 = { "target-features"="+sve" vscale_range(2,16) }
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
new file mode 100644
index 0000000000000..4b62ae7e93f74
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
@@ -0,0 +1,96 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -force-vector-interleave=2 -force-vector-width="vscale x 2" -epilogue-vectorization-force-VF="vscale x 2" -passes=loop-vectorize -mcpu=neoverse-v2 -S %s | FileCheck %s
+
+target triple = "arm64-apple-macosx"
+
+; Test that vectorization does not crash when narrowInterleaveGroups
+; materializes VFxUF and the canonical IV increment step is UF*vscale
+; (in the vector epilogue).
+%pair = type { i64, i64 }
+define void @test(ptr noalias %A, i64 %v, i64 %n) #0 {
+; CHECK-LABEL: define void @test(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[V:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[VECTOR_PH:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK2:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; CHECK-NEXT:    [[TMP2:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[TMP3:%.*]] = shl nuw i64 [[TMP2]], 1
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH1:.*]], label %[[VECTOR_PH1:.*]]
+; CHECK:       [[VECTOR_PH1]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP4]]
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH2:.*]]
+; CHECK:       [[VECTOR_PH2]]:
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[V]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[TMP17:%.*]] = phi i64 [ 0, %[[VECTOR_PH2]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP5:%.*]] = add i64 [[TMP0]], 0
+; CHECK-NEXT:    [[TMP6:%.*]] = mul i64 [[TMP5]], 1
+; CHECK-NEXT:    [[INDEX9:%.*]] = add i64 [[TMP17]], [[TMP6]]
+; CHECK-NEXT:    [[TMP21:%.*]] = getelementptr inbounds [[PAIR:%.*]], ptr [[A]], i64 [[TMP17]], i32 0
+; CHECK-NEXT:    [[TMP24:%.*]] = getelementptr inbounds [[PAIR]], ptr [[A]], i64 [[INDEX9]], i32 0
+; CHECK-NEXT:    store <vscale x 2 x i64> [[BROADCAST_SPLAT]], ptr [[TMP21]], align 8
+; CHECK-NEXT:    store <vscale x 2 x i64> [[BROADCAST_SPLAT]], ptr [[TMP24]], align 8
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[TMP17]], [[TMP1]]
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N1:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N1]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], [[TMP3]]
+; CHECK-NEXT:    br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_PH1]], label %[[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_PH1]] ]
+; CHECK-NEXT:    [[TMP11:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT:    [[N_MOD_VF2:%.*]] = urem i64 [[N]], [[TMP11]]
+; CHECK-NEXT:    [[N_VEC3:%.*]] = sub i64 [[N]], [[N_MOD_VF2]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT4:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[V]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT5:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT4]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT7:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[IDX_0:%.*]] = getelementptr inbounds [[PAIR]], ptr [[A]], i64 [[IV]], i32 0
+; CHECK-NEXT:    store <vscale x 2 x i64> [[BROADCAST_SPLAT5]], ptr [[IDX_0]], align 8
+; CHECK-NEXT:    [[INDEX_NEXT7]] = add nuw i64 [[IV]], [[TMP11]]
+; CHECK-NEXT:    [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT7]], [[N_VEC3]]
+; CHECK-NEXT:    br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK1:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK1]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC3]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT]], label %[[VEC_EPILOG_PH1]]
+; CHECK:       [[VEC_EPILOG_PH1]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC3]], %[[VEC_EPILOG_MIDDLE_BLOCK1]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_PH]] ]
+; CHECK-NEXT:    br label %[[FOR_BODY1:.*]]
+; CHECK:       [[FOR_BODY1]]:
+; CHECK-NEXT:    [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_PH1]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY1]] ]
+; CHECK-NEXT:    [[IDX_2:%.*]] = getelementptr inbounds [[PAIR]], ptr [[A]], i64 [[IV1]], i32 0
+; CHECK-NEXT:    [[IDX_1:%.*]] = getelementptr inbounds [[PAIR]], ptr [[A]], i64 [[IV1]], i32 1
+; CHECK-NEXT:    store i64 [[V]], ptr [[IDX_2]], align 8
+; CHECK-NEXT:    store i64 [[V]], ptr [[IDX_1]], align 8
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
+; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[EXITCOND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+  %idx.0 = getelementptr inbounds %pair, ptr %A, i64 %iv, i32 0
+  %idx.1 = getelementptr inbounds %pair, ptr %A, i64 %iv, i32 1
+  store i64 %v, ptr %idx.0
+  store i64 %v, ptr %idx.1
+  %iv.next = add nuw nsw i64 %iv, 1
+  %exitcond = icmp eq i64 %iv.next, %n
+  br i1 %exitcond, label %exit, label %for.body
+
+exit:
+  ret void
+}
diff --git a/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp b/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
index 7c49487dac459..6b630037d670c 100644
--- a/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
+++ b/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
@@ -1952,37 +1952,5 @@ TEST_F(VPInstructionTest, VPSymbolicValueAddOperandAfterMaterialization) {
 }
 #endif
 
-TEST_F(VPRecipeTest, UFVScaleUserBeforeMaterialization) {
-  VPlan &Plan = getPlan();
-  VPBasicBlock *Header = Plan.createVPBasicBlock("vector.header");
-  VPBasicBlock *Latch = Plan.createVPBasicBlock("vector.latch");
-  VPValue *UF = &Plan.getUF();
-  Type *IVTy = UF->getScalarType();
-  VPRegionBlock *LoopRegion = Plan.createLoopRegion(
-      IVTy, DebugLoc::getUnknown(), "vector.loop", Header, Latch);
-  VPBlockUtils::connectBlocks(Header, Latch);
-  VPBlockUtils::connectBlocks(Plan.getEntry(), LoopRegion);
-  VPBlockUtils::connectBlocks(LoopRegion, Plan.getScalarHeader());
-
-  auto *VScale = VPBuilder(Plan.getVectorPreheader()).createVScale(IVTy);
-
-  auto *Step = new VPInstruction(Instruction::Mul, {VScale, UF},
-                                 VPIRFlags::getDefaultFlags(Instruction::Mul));
-  Plan.getVectorPreheader()->appendRecipe(Step);
-
-  auto *Increment = new VPInstruction(
-      Instruction::Add, {LoopRegion->getCanonicalIV(), Step},
-      VPIRFlags::WrapFlagsTy(LoopRegion->hasCanonicalIVNUW(), false), {},
-      DebugLoc::getUnknown(), "index.next");
-  Latch->appendRecipe(Increment);
-
-  auto *Br = new VPInstruction(VPInstruction::BranchOnCount,
-                               {Increment, &Plan.getVectorTripCount()});
-  Latch->appendRecipe(Br);
-
-  Plan.getVFxUF().markMaterialized();
-  EXPECT_EQ(Increment, LoopRegion->getOrCreateCanonicalIVIncrement());
-}
-
 } // namespace
 } // namespace llvm

>From 2579020b5f5a721b54cdb70c51ebd8957933d1a3 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 22 Jun 2026 13:16:19 +0100
Subject: [PATCH 2/5] Update
 llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll

Co-authored-by: Ramkumar Ramachandra <r at artagnon.com>
---
 .../AArch64/transform-narrow-interleave-vscale-x-UF-step.ll     | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
index 4b62ae7e93f74..2249f23e03f41 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
@@ -1,5 +1,5 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; RUN: opt -force-vector-interleave=2 -force-vector-width="vscale x 2" -epilogue-vectorization-force-VF="vscale x 2" -passes=loop-vectorize -mcpu=neoverse-v2 -S %s | FileCheck %s
+; RUN: opt -passes=loop-vectorize -force-vector-interleave=2 -force-vector-width="vscale x 2" -epilogue-vectorization-force-VF="vscale x 2" -mcpu=neoverse-v2 -S %s | FileCheck %s
 
 target triple = "arm64-apple-macosx"
 

>From 2b6fad6b4025432297acbe8b7f7fe4654647dfc9 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Wed, 24 Jun 2026 15:46:44 +0000
Subject: [PATCH 3/5] Fixups

---
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 14 +++-----------
 1 file changed, 3 insertions(+), 11 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ea143b9799aea..eabdcb7fdaa47 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -179,8 +179,7 @@ static cl::opt<bool> EnableEpilogueVectorization(
     cl::desc("Enable vectorization of epilogue loops."));
 
 static cl::opt<ElementCount> EpilogueVectorizationForceVF(
-    "epilogue-vectorization-force-VF", cl::init(ElementCount::getFixed(1)),
-    cl::Hidden,
+    "epilogue-vectorization-force-VF", cl::Hidden,
     cl::desc("When epilogue vectorization is enabled, and a value greater than "
              "1 is specified, forces the given VF for all applicable epilogue "
              "loops."));
@@ -418,13 +417,6 @@ static cl::opt<bool> EnableEarlyExitVectorizationWithSideEffects(
     cl::desc("Enable vectorization of early exit loops with uncountable exits "
              "and side effects"));
 
-// Returns true if the epilogue VF has been set to a non-zero value other than
-// VF=1 (scalar).
-static bool hasForcedEpilogueVF() {
-  return EpilogueVectorizationForceVF.isNonZero() &&
-         EpilogueVectorizationForceVF != ElementCount::getFixed(1);
-}
-
 // Likelyhood of bypassing the vectorized loop because there are zero trips left
 // after prolog. See `emitIterationCountCheck`.
 static constexpr uint32_t MinItersBypassWeights[] = {1, 127};
@@ -3484,7 +3476,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     return nullptr;
   }
 
-  if (hasForcedEpilogueVF()) {
+  if (EpilogueVectorizationForceVF.isNonZero()) {
     if (estimateElementCount(EpilogueVectorizationForceVF,
                              Config.getVScaleForTuning()) >=
         IC * estimateElementCount(MainLoopVF, Config.getVScaleForTuning())) {
@@ -5809,7 +5801,7 @@ LoopVectorizationPlanner::computeBestVF() {
     return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
   }
 
-  if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF()) {
+  if (hasPlanWithVF(UserVF) && EpilogueVectorizationForceVF.isNonZero()) {
     assert(VPlans.size() == 2 && "Must have exactly 2 VPlans built");
     assert(VPlans[0]->getSingleVF() == EpilogueVectorizationForceVF &&
            "expected first plan to be for the forced epilogue VF");

>From eab1b972a150eab4bd2fa06f25b7409da158dfba Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 29 Jun 2026 10:10:35 +0000
Subject: [PATCH 4/5] Revert "Fixups"

This reverts commit ef9bd8a84aaab99f543fd2c9deb2a090cceb2edd.
---
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 14 +++++++++++---
 1 file changed, 11 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index eabdcb7fdaa47..ea143b9799aea 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -179,7 +179,8 @@ static cl::opt<bool> EnableEpilogueVectorization(
     cl::desc("Enable vectorization of epilogue loops."));
 
 static cl::opt<ElementCount> EpilogueVectorizationForceVF(
-    "epilogue-vectorization-force-VF", cl::Hidden,
+    "epilogue-vectorization-force-VF", cl::init(ElementCount::getFixed(1)),
+    cl::Hidden,
     cl::desc("When epilogue vectorization is enabled, and a value greater than "
              "1 is specified, forces the given VF for all applicable epilogue "
              "loops."));
@@ -417,6 +418,13 @@ static cl::opt<bool> EnableEarlyExitVectorizationWithSideEffects(
     cl::desc("Enable vectorization of early exit loops with uncountable exits "
              "and side effects"));
 
+// Returns true if the epilogue VF has been set to a non-zero value other than
+// VF=1 (scalar).
+static bool hasForcedEpilogueVF() {
+  return EpilogueVectorizationForceVF.isNonZero() &&
+         EpilogueVectorizationForceVF != ElementCount::getFixed(1);
+}
+
 // Likelyhood of bypassing the vectorized loop because there are zero trips left
 // after prolog. See `emitIterationCountCheck`.
 static constexpr uint32_t MinItersBypassWeights[] = {1, 127};
@@ -3476,7 +3484,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     return nullptr;
   }
 
-  if (EpilogueVectorizationForceVF.isNonZero()) {
+  if (hasForcedEpilogueVF()) {
     if (estimateElementCount(EpilogueVectorizationForceVF,
                              Config.getVScaleForTuning()) >=
         IC * estimateElementCount(MainLoopVF, Config.getVScaleForTuning())) {
@@ -5801,7 +5809,7 @@ LoopVectorizationPlanner::computeBestVF() {
     return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
   }
 
-  if (hasPlanWithVF(UserVF) && EpilogueVectorizationForceVF.isNonZero()) {
+  if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF()) {
     assert(VPlans.size() == 2 && "Must have exactly 2 VPlans built");
     assert(VPlans[0]->getSingleVF() == EpilogueVectorizationForceVF &&
            "expected first plan to be for the forced epilogue VF");

>From 933c5566dac3a8eff662a30ca41d54ea1ef70745 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 29 Jun 2026 10:18:11 +0000
Subject: [PATCH 5/5] Fixups

---
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 10 +++++-----
 1 file changed, 5 insertions(+), 5 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ea143b9799aea..c1fbbc2e1ff58 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -183,7 +183,7 @@ static cl::opt<ElementCount> EpilogueVectorizationForceVF(
     cl::Hidden,
     cl::desc("When epilogue vectorization is enabled, and a value greater than "
              "1 is specified, forces the given VF for all applicable epilogue "
-             "loops."));
+             "loops. Note: This allows all scalable VFs >= vscale x 1."));
 
 static cl::opt<unsigned> EpilogueVectorizationMinVF(
     "epilogue-vectorization-minimum-VF", cl::Hidden,
@@ -3497,10 +3497,10 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
     }
 
     LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization factor is forced.\n");
-    ElementCount ForcedEC = EpilogueVectorizationForceVF;
-    if (hasPlanWithVF(ForcedEC)) {
-      std::unique_ptr<VPlan> Clone(getPlanFor(ForcedEC).duplicate());
-      Clone->setVF(ForcedEC);
+    if (hasPlanWithVF(EpilogueVectorizationForceVF)) {
+      std::unique_ptr<VPlan> Clone(
+          getPlanFor(EpilogueVectorizationForceVF).duplicate());
+      Clone->setVF(EpilogueVectorizationForceVF);
       return Clone;
     }
 



More information about the llvm-commits mailing list