[llvm] [LV] Allow setting scalable VFs for `-epilogue-vectorization-force-VF` (PR #205081)
Benjamin Maxwell via llvm-commits
llvm-commits at lists.llvm.org
Tue Jul 7 01:48:45 PDT 2026
https://github.com/MacDue updated https://github.com/llvm/llvm-project/pull/205081
>From 29f3644d4ba97378794c0e390ecb53bfe2f18b8d Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 22 Jun 2026 11:15:10 +0000
Subject: [PATCH 1/5] [LV] Allow setting scalable VFs for
`-epilogue-vectorization-force-VF`
Follow up to #204953. This allows replacing a unit test with an IR test.
---
.../Transforms/Vectorize/LoopVectorize.cpp | 27 ++++--
...interleave-to-widen-memory-epilogue-vec.ll | 26 -----
...form-narrow-interleave-vscale-x-UF-step.ll | 96 +++++++++++++++++++
.../Transforms/Vectorize/VPlanTest.cpp | 32 -------
4 files changed, 113 insertions(+), 68 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 6bb68d1f7bb19..ea143b9799aea 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -178,8 +178,9 @@ static cl::opt<bool> EnableEpilogueVectorization(
"enable-epilogue-vectorization", cl::init(true), cl::Hidden,
cl::desc("Enable vectorization of epilogue loops."));
-static cl::opt<unsigned> EpilogueVectorizationForceVF(
- "epilogue-vectorization-force-VF", cl::init(1), cl::Hidden,
+static cl::opt<ElementCount> EpilogueVectorizationForceVF(
+ "epilogue-vectorization-force-VF", cl::init(ElementCount::getFixed(1)),
+ cl::Hidden,
cl::desc("When epilogue vectorization is enabled, and a value greater than "
"1 is specified, forces the given VF for all applicable epilogue "
"loops."));
@@ -417,6 +418,13 @@ static cl::opt<bool> EnableEarlyExitVectorizationWithSideEffects(
cl::desc("Enable vectorization of early exit loops with uncountable exits "
"and side effects"));
+// Returns true if the epilogue VF has been set to a non-zero value other than
+// VF=1 (scalar).
+static bool hasForcedEpilogueVF() {
+ return EpilogueVectorizationForceVF.isNonZero() &&
+ EpilogueVectorizationForceVF != ElementCount::getFixed(1);
+}
+
// Likelyhood of bypassing the vectorized loop because there are zero trips left
// after prolog. See `emitIterationCountCheck`.
static constexpr uint32_t MinItersBypassWeights[] = {1, 127};
@@ -3476,8 +3484,9 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
return nullptr;
}
- if (EpilogueVectorizationForceVF > 1) {
- if (EpilogueVectorizationForceVF >=
+ if (hasForcedEpilogueVF()) {
+ if (estimateElementCount(EpilogueVectorizationForceVF,
+ Config.getVScaleForTuning()) >=
IC * estimateElementCount(MainLoopVF, Config.getVScaleForTuning())) {
// Note that the main loop leaves IC * MainLoopVF iterations iff a scalar
// epilogue is required, but then the epilogue loop also requires a scalar
@@ -3488,7 +3497,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
}
LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization factor is forced.\n");
- ElementCount ForcedEC = ElementCount::getFixed(EpilogueVectorizationForceVF);
+ ElementCount ForcedEC = EpilogueVectorizationForceVF;
if (hasPlanWithVF(ForcedEC)) {
std::unique_ptr<VPlan> Clone(getPlanFor(ForcedEC).duplicate());
Clone->setVF(ForcedEC);
@@ -5547,8 +5556,7 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
// Collect the instructions (and their associated costs) that will be more
// profitable to scalarize.
CM.collectNonVectorizedAndSetWideningDecisions(UserVF);
- ElementCount EpilogueUserVF =
- ElementCount::getFixed(EpilogueVectorizationForceVF);
+ ElementCount EpilogueUserVF = EpilogueVectorizationForceVF;
if (EpilogueUserVF.isVector() &&
ElementCount::isKnownLT(EpilogueUserVF, UserVF)) {
CM.collectNonVectorizedAndSetWideningDecisions(EpilogueUserVF);
@@ -5801,10 +5809,9 @@ LoopVectorizationPlanner::computeBestVF() {
return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
}
- if (hasPlanWithVF(UserVF) && EpilogueVectorizationForceVF > 1) {
+ if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF()) {
assert(VPlans.size() == 2 && "Must have exactly 2 VPlans built");
- assert(VPlans[0]->getSingleVF() ==
- ElementCount::getFixed(EpilogueVectorizationForceVF) &&
+ assert(VPlans[0]->getSingleVF() == EpilogueVectorizationForceVF &&
"expected first plan to be for the forced epilogue VF");
assert(VPlans[1]->getSingleVF() == UserVF &&
"expected second plan to be for the forced UserVF");
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-epilogue-vec.ll b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-epilogue-vec.ll
index f0b026e7f43a8..bea5fb0ec61d9 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-epilogue-vec.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-to-widen-memory-epilogue-vec.ll
@@ -79,29 +79,3 @@ loop:
exit:
ret i32 0
}
-
-; Test that vectorization does not crash when narrowInterleaveGroups
-; materializes VFxUF and the canonical IV increment step is UF*vscale.
-%pair = type { i64, i64 }
-define void @test(ptr noalias %A, i64 %v, i64 %n) #0 {
-; CHECK-LABEL: define void @test(
-; CHECK: vector.body:
-;
-entry:
- br label %for.body
-
-for.body:
- %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
- %idx.0 = getelementptr inbounds %pair, ptr %A, i64 %iv, i32 0
- %idx.1 = getelementptr inbounds %pair, ptr %A, i64 %iv, i32 1
- store i64 %v, ptr %idx.0
- store i64 %v, ptr %idx.1
- %iv.next = add nuw nsw i64 %iv, 1
- %exitcond = icmp eq i64 %iv.next, %n
- br i1 %exitcond, label %exit, label %for.body
-
-exit:
- ret void
-}
-
-attributes #0 = { "target-features"="+sve" vscale_range(2,16) }
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
new file mode 100644
index 0000000000000..4b62ae7e93f74
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
@@ -0,0 +1,96 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -force-vector-interleave=2 -force-vector-width="vscale x 2" -epilogue-vectorization-force-VF="vscale x 2" -passes=loop-vectorize -mcpu=neoverse-v2 -S %s | FileCheck %s
+
+target triple = "arm64-apple-macosx"
+
+; Test that vectorization does not crash when narrowInterleaveGroups
+; materializes VFxUF and the canonical IV increment step is UF*vscale
+; (in the vector epilogue).
+%pair = type { i64, i64 }
+define void @test(ptr noalias %A, i64 %v, i64 %n) #0 {
+; CHECK-LABEL: define void @test(
+; CHECK-SAME: ptr noalias [[A:%.*]], i64 [[V:%.*]], i64 [[N:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[VECTOR_PH:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 1
+; CHECK-NEXT: [[MIN_ITERS_CHECK2:%.*]] = icmp ult i64 [[N]], [[TMP1]]
+; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP2]], 1
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK2]], label %[[VEC_EPILOG_PH1:.*]], label %[[VECTOR_PH1:.*]]
+; CHECK: [[VECTOR_PH1]]:
+; CHECK-NEXT: [[TMP4:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i64 [[N]], [[TMP4]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH2:.*]]
+; CHECK: [[VECTOR_PH2]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[N]], [[TMP1]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[N]], [[N_MOD_VF]]
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[V]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[TMP17:%.*]] = phi i64 [ 0, %[[VECTOR_PH2]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[TMP0]], 0
+; CHECK-NEXT: [[TMP6:%.*]] = mul i64 [[TMP5]], 1
+; CHECK-NEXT: [[INDEX9:%.*]] = add i64 [[TMP17]], [[TMP6]]
+; CHECK-NEXT: [[TMP21:%.*]] = getelementptr inbounds [[PAIR:%.*]], ptr [[A]], i64 [[TMP17]], i32 0
+; CHECK-NEXT: [[TMP24:%.*]] = getelementptr inbounds [[PAIR]], ptr [[A]], i64 [[INDEX9]], i32 0
+; CHECK-NEXT: store <vscale x 2 x i64> [[BROADCAST_SPLAT]], ptr [[TMP21]], align 8
+; CHECK-NEXT: store <vscale x 2 x i64> [[BROADCAST_SPLAT]], ptr [[TMP24]], align 8
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[TMP17]], [[TMP1]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N1:%.*]] = icmp eq i64 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N1]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK: [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i64 [[N_MOD_VF]], [[TMP3]]
+; CHECK-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_PH1]], label %[[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
+; CHECK: [[VEC_EPILOG_PH]]:
+; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_PH1]] ]
+; CHECK-NEXT: [[TMP11:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[N_MOD_VF2:%.*]] = urem i64 [[N]], [[TMP11]]
+; CHECK-NEXT: [[N_VEC3:%.*]] = sub i64 [[N]], [[N_MOD_VF2]]
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT4:%.*]] = insertelement <vscale x 2 x i64> poison, i64 [[V]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT5:%.*]] = shufflevector <vscale x 2 x i64> [[BROADCAST_SPLATINSERT4]], <vscale x 2 x i64> poison, <vscale x 2 x i32> zeroinitializer
+; CHECK-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK: [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT7:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT: [[IDX_0:%.*]] = getelementptr inbounds [[PAIR]], ptr [[A]], i64 [[IV]], i32 0
+; CHECK-NEXT: store <vscale x 2 x i64> [[BROADCAST_SPLAT5]], ptr [[IDX_0]], align 8
+; CHECK-NEXT: [[INDEX_NEXT7]] = add nuw i64 [[IV]], [[TMP11]]
+; CHECK-NEXT: [[TMP13:%.*]] = icmp eq i64 [[INDEX_NEXT7]], [[N_VEC3]]
+; CHECK-NEXT: br i1 [[TMP13]], label %[[VEC_EPILOG_MIDDLE_BLOCK1:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK1]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N]], [[N_VEC3]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT]], label %[[VEC_EPILOG_PH1]]
+; CHECK: [[VEC_EPILOG_PH1]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC3]], %[[VEC_EPILOG_MIDDLE_BLOCK1]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_PH]] ]
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_PH1]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY1]] ]
+; CHECK-NEXT: [[IDX_2:%.*]] = getelementptr inbounds [[PAIR]], ptr [[A]], i64 [[IV1]], i32 0
+; CHECK-NEXT: [[IDX_1:%.*]] = getelementptr inbounds [[PAIR]], ptr [[A]], i64 [[IV1]], i32 1
+; CHECK-NEXT: store i64 [[V]], ptr [[IDX_2]], align 8
+; CHECK-NEXT: store i64 [[V]], ptr [[IDX_1]], align 8
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %idx.0 = getelementptr inbounds %pair, ptr %A, i64 %iv, i32 0
+ %idx.1 = getelementptr inbounds %pair, ptr %A, i64 %iv, i32 1
+ store i64 %v, ptr %idx.0
+ store i64 %v, ptr %idx.1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, %n
+ br i1 %exitcond, label %exit, label %for.body
+
+exit:
+ ret void
+}
diff --git a/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp b/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
index 7c49487dac459..6b630037d670c 100644
--- a/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
+++ b/llvm/unittests/Transforms/Vectorize/VPlanTest.cpp
@@ -1952,37 +1952,5 @@ TEST_F(VPInstructionTest, VPSymbolicValueAddOperandAfterMaterialization) {
}
#endif
-TEST_F(VPRecipeTest, UFVScaleUserBeforeMaterialization) {
- VPlan &Plan = getPlan();
- VPBasicBlock *Header = Plan.createVPBasicBlock("vector.header");
- VPBasicBlock *Latch = Plan.createVPBasicBlock("vector.latch");
- VPValue *UF = &Plan.getUF();
- Type *IVTy = UF->getScalarType();
- VPRegionBlock *LoopRegion = Plan.createLoopRegion(
- IVTy, DebugLoc::getUnknown(), "vector.loop", Header, Latch);
- VPBlockUtils::connectBlocks(Header, Latch);
- VPBlockUtils::connectBlocks(Plan.getEntry(), LoopRegion);
- VPBlockUtils::connectBlocks(LoopRegion, Plan.getScalarHeader());
-
- auto *VScale = VPBuilder(Plan.getVectorPreheader()).createVScale(IVTy);
-
- auto *Step = new VPInstruction(Instruction::Mul, {VScale, UF},
- VPIRFlags::getDefaultFlags(Instruction::Mul));
- Plan.getVectorPreheader()->appendRecipe(Step);
-
- auto *Increment = new VPInstruction(
- Instruction::Add, {LoopRegion->getCanonicalIV(), Step},
- VPIRFlags::WrapFlagsTy(LoopRegion->hasCanonicalIVNUW(), false), {},
- DebugLoc::getUnknown(), "index.next");
- Latch->appendRecipe(Increment);
-
- auto *Br = new VPInstruction(VPInstruction::BranchOnCount,
- {Increment, &Plan.getVectorTripCount()});
- Latch->appendRecipe(Br);
-
- Plan.getVFxUF().markMaterialized();
- EXPECT_EQ(Increment, LoopRegion->getOrCreateCanonicalIVIncrement());
-}
-
} // namespace
} // namespace llvm
>From 2579020b5f5a721b54cdb70c51ebd8957933d1a3 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 22 Jun 2026 13:16:19 +0100
Subject: [PATCH 2/5] Update
llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
Co-authored-by: Ramkumar Ramachandra <r at artagnon.com>
---
.../AArch64/transform-narrow-interleave-vscale-x-UF-step.ll | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
index 4b62ae7e93f74..2249f23e03f41 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/transform-narrow-interleave-vscale-x-UF-step.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; RUN: opt -force-vector-interleave=2 -force-vector-width="vscale x 2" -epilogue-vectorization-force-VF="vscale x 2" -passes=loop-vectorize -mcpu=neoverse-v2 -S %s | FileCheck %s
+; RUN: opt -passes=loop-vectorize -force-vector-interleave=2 -force-vector-width="vscale x 2" -epilogue-vectorization-force-VF="vscale x 2" -mcpu=neoverse-v2 -S %s | FileCheck %s
target triple = "arm64-apple-macosx"
>From 2b6fad6b4025432297acbe8b7f7fe4654647dfc9 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Wed, 24 Jun 2026 15:46:44 +0000
Subject: [PATCH 3/5] Fixups
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 14 +++-----------
1 file changed, 3 insertions(+), 11 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ea143b9799aea..eabdcb7fdaa47 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -179,8 +179,7 @@ static cl::opt<bool> EnableEpilogueVectorization(
cl::desc("Enable vectorization of epilogue loops."));
static cl::opt<ElementCount> EpilogueVectorizationForceVF(
- "epilogue-vectorization-force-VF", cl::init(ElementCount::getFixed(1)),
- cl::Hidden,
+ "epilogue-vectorization-force-VF", cl::Hidden,
cl::desc("When epilogue vectorization is enabled, and a value greater than "
"1 is specified, forces the given VF for all applicable epilogue "
"loops."));
@@ -418,13 +417,6 @@ static cl::opt<bool> EnableEarlyExitVectorizationWithSideEffects(
cl::desc("Enable vectorization of early exit loops with uncountable exits "
"and side effects"));
-// Returns true if the epilogue VF has been set to a non-zero value other than
-// VF=1 (scalar).
-static bool hasForcedEpilogueVF() {
- return EpilogueVectorizationForceVF.isNonZero() &&
- EpilogueVectorizationForceVF != ElementCount::getFixed(1);
-}
-
// Likelyhood of bypassing the vectorized loop because there are zero trips left
// after prolog. See `emitIterationCountCheck`.
static constexpr uint32_t MinItersBypassWeights[] = {1, 127};
@@ -3484,7 +3476,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
return nullptr;
}
- if (hasForcedEpilogueVF()) {
+ if (EpilogueVectorizationForceVF.isNonZero()) {
if (estimateElementCount(EpilogueVectorizationForceVF,
Config.getVScaleForTuning()) >=
IC * estimateElementCount(MainLoopVF, Config.getVScaleForTuning())) {
@@ -5809,7 +5801,7 @@ LoopVectorizationPlanner::computeBestVF() {
return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
}
- if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF()) {
+ if (hasPlanWithVF(UserVF) && EpilogueVectorizationForceVF.isNonZero()) {
assert(VPlans.size() == 2 && "Must have exactly 2 VPlans built");
assert(VPlans[0]->getSingleVF() == EpilogueVectorizationForceVF &&
"expected first plan to be for the forced epilogue VF");
>From eab1b972a150eab4bd2fa06f25b7409da158dfba Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 29 Jun 2026 10:10:35 +0000
Subject: [PATCH 4/5] Revert "Fixups"
This reverts commit ef9bd8a84aaab99f543fd2c9deb2a090cceb2edd.
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 14 +++++++++++---
1 file changed, 11 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index eabdcb7fdaa47..ea143b9799aea 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -179,7 +179,8 @@ static cl::opt<bool> EnableEpilogueVectorization(
cl::desc("Enable vectorization of epilogue loops."));
static cl::opt<ElementCount> EpilogueVectorizationForceVF(
- "epilogue-vectorization-force-VF", cl::Hidden,
+ "epilogue-vectorization-force-VF", cl::init(ElementCount::getFixed(1)),
+ cl::Hidden,
cl::desc("When epilogue vectorization is enabled, and a value greater than "
"1 is specified, forces the given VF for all applicable epilogue "
"loops."));
@@ -417,6 +418,13 @@ static cl::opt<bool> EnableEarlyExitVectorizationWithSideEffects(
cl::desc("Enable vectorization of early exit loops with uncountable exits "
"and side effects"));
+// Returns true if the epilogue VF has been set to a non-zero value other than
+// VF=1 (scalar).
+static bool hasForcedEpilogueVF() {
+ return EpilogueVectorizationForceVF.isNonZero() &&
+ EpilogueVectorizationForceVF != ElementCount::getFixed(1);
+}
+
// Likelyhood of bypassing the vectorized loop because there are zero trips left
// after prolog. See `emitIterationCountCheck`.
static constexpr uint32_t MinItersBypassWeights[] = {1, 127};
@@ -3476,7 +3484,7 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
return nullptr;
}
- if (EpilogueVectorizationForceVF.isNonZero()) {
+ if (hasForcedEpilogueVF()) {
if (estimateElementCount(EpilogueVectorizationForceVF,
Config.getVScaleForTuning()) >=
IC * estimateElementCount(MainLoopVF, Config.getVScaleForTuning())) {
@@ -5801,7 +5809,7 @@ LoopVectorizationPlanner::computeBestVF() {
return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
}
- if (hasPlanWithVF(UserVF) && EpilogueVectorizationForceVF.isNonZero()) {
+ if (hasPlanWithVF(UserVF) && hasForcedEpilogueVF()) {
assert(VPlans.size() == 2 && "Must have exactly 2 VPlans built");
assert(VPlans[0]->getSingleVF() == EpilogueVectorizationForceVF &&
"expected first plan to be for the forced epilogue VF");
>From 933c5566dac3a8eff662a30ca41d54ea1ef70745 Mon Sep 17 00:00:00 2001
From: Benjamin Maxwell <benjamin.maxwell at arm.com>
Date: Mon, 29 Jun 2026 10:18:11 +0000
Subject: [PATCH 5/5] Fixups
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 10 +++++-----
1 file changed, 5 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ea143b9799aea..c1fbbc2e1ff58 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -183,7 +183,7 @@ static cl::opt<ElementCount> EpilogueVectorizationForceVF(
cl::Hidden,
cl::desc("When epilogue vectorization is enabled, and a value greater than "
"1 is specified, forces the given VF for all applicable epilogue "
- "loops."));
+ "loops. Note: This allows all scalable VFs >= vscale x 1."));
static cl::opt<unsigned> EpilogueVectorizationMinVF(
"epilogue-vectorization-minimum-VF", cl::Hidden,
@@ -3497,10 +3497,10 @@ std::unique_ptr<VPlan> LoopVectorizationPlanner::selectBestEpiloguePlan(
}
LLVM_DEBUG(dbgs() << "LEV: Epilogue vectorization factor is forced.\n");
- ElementCount ForcedEC = EpilogueVectorizationForceVF;
- if (hasPlanWithVF(ForcedEC)) {
- std::unique_ptr<VPlan> Clone(getPlanFor(ForcedEC).duplicate());
- Clone->setVF(ForcedEC);
+ if (hasPlanWithVF(EpilogueVectorizationForceVF)) {
+ std::unique_ptr<VPlan> Clone(
+ getPlanFor(EpilogueVectorizationForceVF).duplicate());
+ Clone->setVF(EpilogueVectorizationForceVF);
return Clone;
}
More information about the llvm-commits
mailing list