[llvm] [LV] Always try to use fixed VF for low TC loops if no epilogue allowed. (PR #226318)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 25 23:04:49 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/226318
>From 8f6e9beada60654cc040bac949af8060209703d6 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 23 Sep 2026 10:59:00 +0100
Subject: [PATCH 1/3] [LV] Always try to use fixed VF for low TC loops if no
epilogue allowed.
Remove the MaxPowerOf2RuntimeVF gate for falling back to using fixed
width VFs when no epilogue is allowed and the trip count is below the
minimum for tail folding.
MaxPowerOf2RuntimeVF being not set means we cannot compute the maximum
runtime VF, due to missing max vscale. In that case we are not able to
determine if a epilogue loop remains for scalable VFs, but we can still
pick a fixed VF, if no epilogue remains for it.
Also updates processLoop to ignore CM_EpilogueNotNeededFoldTail, if the
trip count is below the tail-folding threshold.
---
.../Transforms/Vectorize/LoopVectorize.cpp | 31 +++++++------
.../LoopVectorize/RISCV/short-trip-count.ll | 44 +++++++++----------
2 files changed, 39 insertions(+), 36 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 2f1fc4398654ae..78069c5f985bd5 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3056,17 +3056,16 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
if (ExpectedTC && ExpectedTC->isFixed() &&
ExpectedTC->getFixedValue() <=
TTI.getMinTripCountTailFoldingThreshold()) {
- if (MaxPowerOf2RuntimeVF > 0u) {
- // If we have a low-trip-count, and the fixed-width VF is known to divide
- // the trip count but the scalable factor does not, use the fixed-width
- // factor in preference to allow the generation of a non-predicated loop.
- if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
- NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue())) {
- LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width so that no tail will "
- "remain for any chosen VF.\n");
- MaxFactors.ScalableVF = ElementCount::getScalable(0);
- return MaxFactors;
- }
+ // If we have a low-trip-count, and the fixed-width VF is known to divide
+ // the trip count but the scalable factor does not (or its maximum runtime
+ // VF is unknown), use the fixed-width factor in preference to allow the
+ // generation of a non-predicated loop.
+ if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
+ NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue())) {
+ LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width so that no tail will "
+ "remain for any chosen VF.\n");
+ MaxFactors.ScalableVF = ElementCount::getScalable(0);
+ return MaxFactors;
}
// Allow cases where the ExactTC == (VF * IC) or ExactTC == (VF * IC) + 1.
@@ -3092,7 +3091,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
if (NumOfInstructions > LowTripCountLoopBodySizeLimit) {
unsigned VF = MaxVFForTC / EffectiveIC;
LLVM_DEBUG(dbgs() << "LV: Picking MaxVF=" << VF
- << " with at most 1 scalar iteration remaining.\n");
+ << " with 1 scalar iteration remaining.\n");
MaxFactors.FixedVF = ElementCount::getFixed(VF);
MaxFactors.ScalableVF = ElementCount::getScalable(0);
return MaxFactors;
@@ -7813,8 +7812,12 @@ bool LoopVectorizePass::processLoop(Loop *L) {
// `CM_EpilogueNotAllowedLowTripLoop` prevents vectorizing loops
// with runtime checks. It's more effective to let
// `isOutsideLoopWorkProfitable` determine if vectorization is
- // beneficial for the loop.
- if (SEL != CM_EpilogueNotNeededFoldTail)
+ // beneficial for the loop. If the trip count is below the target's
+ // minimum for tail-folding, the tail cannot be folded, so treat it like
+ // any other low trip count loop.
+ if (SEL != CM_EpilogueNotNeededFoldTail ||
+ ExpectedTC->getFixedValue() <=
+ TTI->getMinTripCountTailFoldingThreshold())
SEL = CM_EpilogueNotAllowedLowTripLoop;
}
}
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/short-trip-count.ll b/llvm/test/Transforms/LoopVectorize/RISCV/short-trip-count.ll
index b370fbe2da7775..8a7fed496e6a3f 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/short-trip-count.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/short-trip-count.ll
@@ -4,17 +4,17 @@
define void @small_trip_count_min_vlen_128(ptr nocapture %a) vscale_range(4,1024) {
; CHECK-LABEL: define void @small_trip_count_min_vlen_128(
; CHECK-SAME: ptr captures(none) [[A:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[ENTRY:.*]]:
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[LOOP]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[IV]]
-; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[GEP]], align 4
-; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[V]], 1
-; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP]], align 4
-; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
-; CHECK-NEXT: [[COND:%.*]] = icmp eq i32 [[IV]], 3
-; CHECK-NEXT: br i1 [[COND]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[A]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -38,17 +38,17 @@ exit:
define void @small_trip_count_min_vlen_32(ptr nocapture %a) vscale_range(1,1024) {
; CHECK-LABEL: define void @small_trip_count_min_vlen_32(
; CHECK-SAME: ptr captures(none) [[A:%.*]]) #[[ATTR1:[0-9]+]] {
-; CHECK-NEXT: [[ENTRY:.*]]:
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[LOOP]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[IV]]
-; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[GEP]], align 4
-; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[V]], 1
-; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP]], align 4
-; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
-; CHECK-NEXT: [[COND:%.*]] = icmp eq i32 [[IV]], 3
-; CHECK-NEXT: br i1 [[COND]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[A]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
>From 336859fa4394841cdd422c280b3b77ec1d0d0af0 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Fri, 25 Sep 2026 18:53:10 +0100
Subject: [PATCH 2/3] !fixup add test, remove stray changea
---
.../Transforms/Vectorize/LoopVectorize.cpp | 2 +-
.../low-trip-count-fixed-vf-no-epilogue.ll | 108 ++++++++++++++++++
2 files changed, 109 insertions(+), 1 deletion(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/low-trip-count-fixed-vf-no-epilogue.ll
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 78069c5f985bd5..cc0c8041cae1e5 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3091,7 +3091,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
if (NumOfInstructions > LowTripCountLoopBodySizeLimit) {
unsigned VF = MaxVFForTC / EffectiveIC;
LLVM_DEBUG(dbgs() << "LV: Picking MaxVF=" << VF
- << " with 1 scalar iteration remaining.\n");
+ << " with at most 1 scalar iteration remaining.\n");
MaxFactors.FixedVF = ElementCount::getFixed(VF);
MaxFactors.ScalableVF = ElementCount::getScalable(0);
return MaxFactors;
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/low-trip-count-fixed-vf-no-epilogue.ll b/llvm/test/Transforms/LoopVectorize/AArch64/low-trip-count-fixed-vf-no-epilogue.ll
new file mode 100644
index 00000000000000..925fa172672496
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/low-trip-count-fixed-vf-no-epilogue.ll
@@ -0,0 +1,108 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve -S %s | FileCheck %s
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64 -mattr=+sve -tail-folding-policy=prefer-fold-tail -S %s | FileCheck %s
+
+; Trip count 4 is below the SVE tail-folding threshold. Pick a fixed-width VF
+; that divides the trip count.
+define void @tc4_no_vscale_range(ptr %a) {
+; CHECK-LABEL: define void @tc4_no_vscale_range(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[A]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep = getelementptr inbounds i32, ptr %a, i64 %iv
+ %v = load i32, ptr %gep, align 4
+ %add = add nsw i32 %v, 1
+ store i32 %add, ptr %gep, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 4
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @tc4_vscale_range(ptr %a) vscale_range(1,16) {
+; CHECK-LABEL: define void @tc4_vscale_range(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[A]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep = getelementptr inbounds i32, ptr %a, i64 %iv
+ %v = load i32, ptr %gep, align 4
+ %add = add nsw i32 %v, 1
+ store i32 %add, ptr %gep, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 4
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+; Trip count 3 is not a multiple of any fixed-width VF.
+define void @tc3_no_vscale_range(ptr %a) {
+; CHECK-LABEL: define void @tc3_no_vscale_range(
+; CHECK-SAME: ptr [[A:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[V]], 1
+; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 3
+; CHECK-NEXT: br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep = getelementptr inbounds i32, ptr %a, i64 %iv
+ %v = load i32, ptr %gep, align 4
+ %add = add nsw i32 %v, 1
+ store i32 %add, ptr %gep, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 3
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
>From 972f476c315bea7be48754952c1e601f00a78ffa Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Fri, 25 Sep 2026 18:57:00 +0100
Subject: [PATCH 3/3] !fixup adjust comment thanks
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 3 +--
1 file changed, 1 insertion(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index cc0c8041cae1e5..f8e802a8cd0a12 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3057,8 +3057,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
ExpectedTC->getFixedValue() <=
TTI.getMinTripCountTailFoldingThreshold()) {
// If we have a low-trip-count, and the fixed-width VF is known to divide
- // the trip count but the scalable factor does not (or its maximum runtime
- // VF is unknown), use the fixed-width factor in preference to allow the
+ // the trip count the fixed-width factor in preference to allow the
// generation of a non-predicated loop.
if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue())) {
More information about the llvm-commits
mailing list