[llvm] [LoopVectorize] Improve Vectorization of Low Trip Count Loops (PR #195823)
Jack Styles via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 02:24:54 PDT 2026
https://github.com/Stylie777 updated https://github.com/llvm/llvm-project/pull/195823
>From a774e82ff914124a8f40db5f2d0e19439444edda Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Fri, 1 May 2026 11:43:47 +0100
Subject: [PATCH 01/32] [LoopVectorize] Improve Vectorization of Small Loops
Currently, Small Loops with Trip Counts less than 16, and in
situations where the Trip Count (TC) is less than the Tail Folding
Threshold are harder to vectorize, its only possible where no epilogue
will be emitted. However, for loops with large bodies and small trip
counts this can be counterprodictive to performance, often failing to
vectorize entirely. This is more prevelant with targets where
`getMinTripCountTailFoldingThreshold()` returns a value greater than 0.
To address this, the Small Loops where the TC == VF + 1 can now vectorize,
leading to a single vectorized itneration and a single scalar iteration.
Later passes can then remove the loop's entirely.
Testing an with OpenSource Fortran HPC Benchmark which includes multiple
loops with small trip counts, but large loop bodies, has shown significant
improvement to runtime after these changes.
Assisted-by: Claude Sonnet 4.6/Codex
---
.../Transforms/Vectorize/LoopVectorize.cpp | 38 +++-
.../AArch64/sve-low-trip-count.ll | 24 +--
.../sve-small-trip-count-vf-plus-one.ll | 195 ++++++++++++++++++
3 files changed, 243 insertions(+), 14 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 24083eb7e8e6e..0a6c848ff873a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3052,6 +3052,15 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
}
auto ExpectedTC = getSmallBestKnownTC(PSE, TheLoop);
+ auto ApplyVectorWidth = [](FixedScalableVFPair &MaxFactors,
+ unsigned int FixedVF, unsigned int ScalableVF) {
+ MaxFactors.FixedVF = ElementCount::getFixed(FixedVF);
+ MaxFactors.ScalableVF = ElementCount::getScalable(ScalableVF);
+ };
+ unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
+ auto HasOneScalarIterationRemainder = [EffectiveIC](ElementCount &ExactTC, unsigned int VF)-> bool {
+ return ExactTC.getFixedValue() == ((VF * EffectiveIC) + 1);
+ };
if (ExpectedTC && ExpectedTC->isFixed() &&
ExpectedTC->getFixedValue() <=
TTI.getMinTripCountTailFoldingThreshold()) {
@@ -3063,9 +3072,36 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue())) {
LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width so that no tail will "
"remain for any chosen VF.\n");
- MaxFactors.ScalableVF = ElementCount::getScalable(0);
+ ApplyVectorWidth(MaxFactors, MaxFactors.FixedVF.getFixedValue(), 0);
return MaxFactors;
}
+ // Allow cases where the ExactTC == VF + 1. VF can be any power of
+ // 2 between 2 and MaxVF.
+ //
+ // This produces 1 vector iteration, and 1 scalar iteration with
+ // no remainder. Later passes will eliminate the loop and leave
+ // straight-line code as the both iteration counts are statically known.
+ ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), TheLoop);
+ if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
+ ExactTC && ExactTC.isFixed()) {
+ if (HasOneScalarIterationRemainder(ExactTC, MaxFactors.FixedVF.getFixedValue())) {
+ LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width with 1 scalar "
+ "iteration remainder.\n");
+ ApplyVectorWidth(MaxFactors, MaxFactors.FixedVF.getFixedValue(), 0);
+ return MaxFactors;
+ }
+ // If the maximum VF cannot produce 1 vector iteration + 1 scalar
+ // iteration, step down VF's to find one that can.
+ for (unsigned VF = MaxFactors.FixedVF.getFixedValue(); VF >= 2;
+ VF /= 2) {
+ if (HasOneScalarIterationRemainder(ExactTC, VF)) {
+ LLVM_DEBUG(dbgs() << "LV: Picking VF=" << VF
+ << " with 1 scalar iteration remainder.\n");
+ ApplyVectorWidth(MaxFactors, VF, 0);
+ return MaxFactors;
+ }
+ }
+ }
}
reportVectorizationFailure(
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll
index c36daa40f6193..76642c63dbd26 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll
@@ -56,22 +56,20 @@ exit:
define void @trip5_i8(ptr noalias nocapture noundef %dst, ptr noalias nocapture noundef readonly %src) #0 {
; CHECK-LABEL: define void @trip5_i8(
; CHECK-SAME: ptr noalias noundef captures(none) [[DST:%.*]], ptr noalias noundef readonly captures(none) [[SRC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[GEP_SRC:%.*]] = getelementptr inbounds i8, ptr [[SRC]], i64 [[IV]]
-; CHECK-NEXT: [[TMP0:%.*]] = load i8, ptr [[GEP_SRC]], align 1
-; CHECK-NEXT: [[MUL:%.*]] = shl i8 [[TMP0]], 1
-; CHECK-NEXT: [[GEP_DST:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 [[IV]]
-; CHECK-NEXT: [[TMP1:%.*]] = load i8, ptr [[GEP_DST]], align 1
-; CHECK-NEXT: [[ADD:%.*]] = add i8 [[MUL]], [[TMP1]]
-; CHECK-NEXT: store i8 [[ADD]], ptr [[GEP_DST]], align 1
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-NEXT: [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK-NEXT: br label %[[EXIT:.*]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: ret void
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[SRC]], align 1
+; CHECK-NEXT: [[TMP0:%.*]] = shl <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[DST]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = add <4 x i8> [[TMP0]], [[WIDE_LOAD1]]
+; CHECK-NEXT: store <4 x i8> [[TMP1]], ptr [[DST]], align 1
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
;
entry:
br label %loop
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
new file mode 100644
index 0000000000000..71157580334d1
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -0,0 +1,195 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+;
+; Test that a loop with trip count == VF + 1 is allowed to vectorize
+; on AArch64+SVE where getMinTripCountTailFoldingThreshold() returns 5. This
+; produces one vector iteration and one scalar iteration.
+;
+; RUN: opt -S -p loop-vectorize %s | FileCheck %s
+
+target triple = "aarch64-unknown-linux-gnu"
+
+; TC=5, VF=4: TC == MaxFixedVF + 1 (5 == 4 + 1).
+; The new code path should trigger: 1 vectorized iteration + 1 scalar iteration.
+define void @tc5_vf4_vectorize(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc5_vf4_vectorize(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[B]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[VAL]], 1
+; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ %val = load i32, ptr %gep.a, align 4
+ %add = add nsw i32 %val, 1
+ store i32 %add, ptr %gep.b, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop
+exit:
+ ret void
+}
+
+; TC=5, VF=2, IC=2: TC == VF * IC + 1 (5 == 2 * 2 + 1).
+; The forced interleave count should be considered when choosing VF.
+define void @tc5_forced_ic2_vectorize(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc5_forced_ic2_vectorize(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 2
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <2 x i32>, ptr [[TMP0]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = add nsw <2 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: [[TMP2:%.*]] = add nsw <2 x i32> [[WIDE_LOAD1]], splat (i32 1)
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 2
+; CHECK-NEXT: store <2 x i32> [[TMP1]], ptr [[B]], align 4
+; CHECK-NEXT: store <2 x i32> [[TMP2]], ptr [[TMP3]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[VAL]], 1
+; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ %val = load i32, ptr %gep.a, align 4
+ %add = add nsw i32 %val, 1
+ store i32 %add, ptr %gep.b, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop, !llvm.loop !0
+exit:
+ ret void
+}
+
+; TC=3, VF=2: TC == FixedVF + 1 (3 == 2 + 1).
+; Should vectorize: 1 vector iteration of width 2, then 1 scalar iteration.
+define void @tc3_vf2_vectorize(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc3_vf2_vectorize(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <2 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <2 x i32> [[TMP0]], ptr [[B]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 2, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[VAL]], 1
+; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 3
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ %val = load i32, ptr %gep.a, align 4
+ %add = add nsw i32 %val, 1
+ store i32 %add, ptr %gep.b, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop
+exit:
+ ret void
+}
+
+; TC=4: exact multiple of VF=4. Vectorizes via the original
+; "no scalar epilogue needed" path -- NOT the new TC==VF+1 path.
+define void @tc4_vf4_vectorize(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc4_vf4_vectorize(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[B]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ %val = load i32, ptr %gep.a, align 4
+ %add = add nsw i32 %val, 1
+ store i32 %add, ptr %gep.b, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 4
+ br i1 %exitcond, label %exit, label %loop
+exit:
+ ret void
+}
+
+attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.interleave.count", i32 2}
>From 7b92a42ea7c5bcf21f750aa5e99dd6f06efc0174 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Tue, 5 May 2026 11:29:55 +0100
Subject: [PATCH 02/32] formatting
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 6 ++++--
1 file changed, 4 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 0a6c848ff873a..50a8edbd33783 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3058,7 +3058,8 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
MaxFactors.ScalableVF = ElementCount::getScalable(ScalableVF);
};
unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
- auto HasOneScalarIterationRemainder = [EffectiveIC](ElementCount &ExactTC, unsigned int VF)-> bool {
+ auto HasOneScalarIterationRemainder = [EffectiveIC](ElementCount &ExactTC,
+ unsigned int VF) -> bool {
return ExactTC.getFixedValue() == ((VF * EffectiveIC) + 1);
};
if (ExpectedTC && ExpectedTC->isFixed() &&
@@ -3084,7 +3085,8 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), TheLoop);
if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
ExactTC && ExactTC.isFixed()) {
- if (HasOneScalarIterationRemainder(ExactTC, MaxFactors.FixedVF.getFixedValue())) {
+ if (HasOneScalarIterationRemainder(
+ ExactTC, MaxFactors.FixedVF.getFixedValue())) {
LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width with 1 scalar "
"iteration remainder.\n");
ApplyVectorWidth(MaxFactors, MaxFactors.FixedVF.getFixedValue(), 0);
>From 4aa6b5e63dc9cb4054c6691954c6afe005ac885e Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Tue, 5 May 2026 15:50:14 +0100
Subject: [PATCH 03/32] Responding to review comments
---
.../Transforms/Vectorize/LoopVectorize.cpp | 38 +++---
.../AArch64/sve-low-trip-count.ll | 38 ------
.../sve-small-trip-count-vf-plus-one.ll | 86 +-------------
.../RISCV/small-trip-count-vf-plus-one.ll | 109 ++++++++++++++++++
4 files changed, 127 insertions(+), 144 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/RISCV/small-trip-count-vf-plus-one.ll
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 50a8edbd33783..ccf4e157b4cfe 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3052,15 +3052,10 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
}
auto ExpectedTC = getSmallBestKnownTC(PSE, TheLoop);
- auto ApplyVectorWidth = [](FixedScalableVFPair &MaxFactors,
- unsigned int FixedVF, unsigned int ScalableVF) {
- MaxFactors.FixedVF = ElementCount::getFixed(FixedVF);
- MaxFactors.ScalableVF = ElementCount::getScalable(ScalableVF);
- };
unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
auto HasOneScalarIterationRemainder = [EffectiveIC](ElementCount &ExactTC,
- unsigned int VF) -> bool {
- return ExactTC.getFixedValue() == ((VF * EffectiveIC) + 1);
+ unsigned int MaxVF) -> bool {
+ return ExactTC.getFixedValue() == 1 + (MaxVF * EffectiveIC);
};
if (ExpectedTC && ExpectedTC->isFixed() &&
ExpectedTC->getFixedValue() <=
@@ -3073,7 +3068,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue())) {
LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width so that no tail will "
"remain for any chosen VF.\n");
- ApplyVectorWidth(MaxFactors, MaxFactors.FixedVF.getFixedValue(), 0);
+ MaxFactors.ScalableVF = ElementCount::getScalable(0);
return MaxFactors;
}
// Allow cases where the ExactTC == VF + 1. VF can be any power of
@@ -3084,22 +3079,21 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// straight-line code as the both iteration counts are statically known.
ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), TheLoop);
if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
- ExactTC && ExactTC.isFixed()) {
- if (HasOneScalarIterationRemainder(
- ExactTC, MaxFactors.FixedVF.getFixedValue())) {
- LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width with 1 scalar "
- "iteration remainder.\n");
- ApplyVectorWidth(MaxFactors, MaxFactors.FixedVF.getFixedValue(), 0);
- return MaxFactors;
- }
+ ExactTC.isFixed()) {
// If the maximum VF cannot produce 1 vector iteration + 1 scalar
- // iteration, step down VF's to find one that can.
- for (unsigned VF = MaxFactors.FixedVF.getFixedValue(); VF >= 2;
- VF /= 2) {
- if (HasOneScalarIterationRemainder(ExactTC, VF)) {
- LLVM_DEBUG(dbgs() << "LV: Picking VF=" << VF
+ // iteration, step down VF's to find one that can. The result should
+ // also eliminate any loops.
+ //
+ // Forced interleaving is considered when seeing if OneScalarIterationRemainder
+ // is produced. It may prodiced more than one vector iteration, but only one
+ // scalar iteration.
+ for (unsigned MaxVF = MaxFactors.FixedVF.getFixedValue(); MaxVF >= 2;
+ MaxVF /= 2) {
+ if (HasOneScalarIterationRemainder(ExactTC, MaxVF)) {
+ LLVM_DEBUG(dbgs() << "LV: Picking VF=" << MaxVF
<< " with 1 scalar iteration remainder.\n");
- ApplyVectorWidth(MaxFactors, VF, 0);
+ MaxFactors.FixedVF = ElementCount::getFixed(MaxVF);
+ MaxFactors.ScalableVF = ElementCount::getScalable(0);
return MaxFactors;
}
}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll
index 76642c63dbd26..03746ee82223a 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll
@@ -53,42 +53,4 @@ exit:
ret void
}
-define void @trip5_i8(ptr noalias nocapture noundef %dst, ptr noalias nocapture noundef readonly %src) #0 {
-; CHECK-LABEL: define void @trip5_i8(
-; CHECK-SAME: ptr noalias noundef captures(none) [[DST:%.*]], ptr noalias noundef readonly captures(none) [[SRC:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: br label %[[EXIT:.*]]
-; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[SRC]], align 1
-; CHECK-NEXT: [[TMP0:%.*]] = shl <4 x i8> [[WIDE_LOAD]], splat (i8 1)
-; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[DST]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = add <4 x i8> [[TMP0]], [[WIDE_LOAD1]]
-; CHECK-NEXT: store <4 x i8> [[TMP1]], ptr [[DST]], align 1
-; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
-; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
-; CHECK: [[SCALAR_PH]]:
-;
-entry:
- br label %loop
-
-loop:
- %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
- %gep.src = getelementptr inbounds i8, ptr %src, i64 %iv
- %0 = load i8, ptr %gep.src, align 1
- %mul = shl i8 %0, 1
- %gep.dst = getelementptr inbounds i8, ptr %dst, i64 %iv
- %1 = load i8, ptr %gep.dst, align 1
- %add = add i8 %mul, %1
- store i8 %add, ptr %gep.dst, align 1
- %iv.next = add nuw nsw i64 %iv, 1
- %ec = icmp eq i64 %iv.next, 5
- br i1 %ec, label %exit, label %loop
-
-exit:
- ret void
-}
-
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
index 71157580334d1..711ce3bc439d2 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -1,8 +1,8 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
;
; Test that a loop with trip count == VF + 1 is allowed to vectorize
-; on AArch64+SVE where getMinTripCountTailFoldingThreshold() returns 5. This
-; produces one vector iteration and one scalar iteration.
+; on AArch64 where under getMinTripCountTailFoldingThreshold(). This
+; produces the required number of vector iteration and one scalar iteration.
;
; RUN: opt -S -p loop-vectorize %s | FileCheck %s
@@ -107,88 +107,6 @@ exit:
ret void
}
-; TC=3, VF=2: TC == FixedVF + 1 (3 == 2 + 1).
-; Should vectorize: 1 vector iteration of width 2, then 1 scalar iteration.
-define void @tc3_vf2_vectorize(ptr noalias %a, ptr noalias %b) #0 {
-; CHECK-LABEL: define void @tc3_vf2_vectorize(
-; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
-; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i32>, ptr [[A]], align 4
-; CHECK-NEXT: [[TMP0:%.*]] = add nsw <2 x i32> [[WIDE_LOAD]], splat (i32 1)
-; CHECK-NEXT: store <2 x i32> [[TMP0]], ptr [[B]], align 4
-; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
-; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 2, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
-; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
-; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
-; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
-; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[VAL]], 1
-; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 3
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP4:![0-9]+]]
-; CHECK: [[EXIT]]:
-; CHECK-NEXT: ret void
-;
-entry:
- br label %loop
-loop:
- %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
- %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
- %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
- %val = load i32, ptr %gep.a, align 4
- %add = add nsw i32 %val, 1
- store i32 %add, ptr %gep.b, align 4
- %iv.next = add nuw nsw i64 %iv, 1
- %exitcond = icmp eq i64 %iv.next, 3
- br i1 %exitcond, label %exit, label %loop
-exit:
- ret void
-}
-
-; TC=4: exact multiple of VF=4. Vectorizes via the original
-; "no scalar epilogue needed" path -- NOT the new TC==VF+1 path.
-define void @tc4_vf4_vectorize(ptr noalias %a, ptr noalias %b) #0 {
-; CHECK-LABEL: define void @tc4_vf4_vectorize(
-; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
-; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
-; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
-; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
-; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[B]], align 4
-; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
-; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[EXIT:.*]]
-; CHECK: [[EXIT]]:
-; CHECK-NEXT: ret void
-;
-entry:
- br label %loop
-loop:
- %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
- %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
- %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
- %val = load i32, ptr %gep.a, align 4
- %add = add nsw i32 %val, 1
- store i32 %add, ptr %gep.b, align 4
- %iv.next = add nuw nsw i64 %iv, 1
- %exitcond = icmp eq i64 %iv.next, 4
- br i1 %exitcond, label %exit, label %loop
-exit:
- ret void
-}
-
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
!0 = distinct !{!0, !1}
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/RISCV/small-trip-count-vf-plus-one.ll
new file mode 100644
index 0000000000000..39917f15e2870
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/small-trip-count-vf-plus-one.ll
@@ -0,0 +1,109 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+;
+; Test that a loop with trip count == VF + 1 is allowed to vectorize
+; on RISCV where under getMinTripCountTailFoldingThreshold(). This
+; produces the required number of vector iteration and one scalar iteration.
+;
+; RUN: opt -S -p loop-vectorize %s -mtriple=riscv64 -mattr=+v -tail-folding-policy=dont-fold-tail | FileCheck %s
+
+; TC=5, VF=4: TC == MaxFixedVF + 1 (5 == 4 + 1).
+; The new code path should trigger: 1 vectorized iteration + 1 scalar iteration.
+define void @tc5_vf4_vectorize(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc5_vf4_vectorize(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[SCALAR_PH1:.*:]]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[B]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[VAL]], 1
+; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ %val = load i32, ptr %gep.a, align 4
+ %add = add nsw i32 %val, 1
+ store i32 %add, ptr %gep.b, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop
+exit:
+ ret void
+}
+
+; TC=5, VF=2, IC=2: TC == VF * IC + 1 (5 == 2 * 2 + 1).
+; The forced interleave count should be considered when choosing VF.
+define void @tc5_forced_ic2_vectorize(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc5_forced_ic2_vectorize(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[SCALAR_PH1:.*:]]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 2
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <2 x i32>, ptr [[TMP0]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = add nsw <2 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: [[TMP2:%.*]] = add nsw <2 x i32> [[WIDE_LOAD1]], splat (i32 1)
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 2
+; CHECK-NEXT: store <2 x i32> [[TMP1]], ptr [[B]], align 4
+; CHECK-NEXT: store <2 x i32> [[TMP2]], ptr [[TMP3]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[VAL]], 1
+; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ %val = load i32, ptr %gep.a, align 4
+ %add = add nsw i32 %val, 1
+ store i32 %add, ptr %gep.b, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop, !llvm.loop !0
+exit:
+ ret void
+}
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.interleave.count", i32 2}
>From 337fe25ddfd93f9e03f90151550a9ca191654b45 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Tue, 5 May 2026 16:30:20 +0100
Subject: [PATCH 04/32] formatting
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 12 ++++++------
1 file changed, 6 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ccf4e157b4cfe..a07e21e99c46a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3053,8 +3053,8 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
auto ExpectedTC = getSmallBestKnownTC(PSE, TheLoop);
unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
- auto HasOneScalarIterationRemainder = [EffectiveIC](ElementCount &ExactTC,
- unsigned int MaxVF) -> bool {
+ auto HasOneScalarIterationRemainder =
+ [EffectiveIC](ElementCount &ExactTC, unsigned int MaxVF) -> bool {
return ExactTC.getFixedValue() == 1 + (MaxVF * EffectiveIC);
};
if (ExpectedTC && ExpectedTC->isFixed() &&
@@ -3083,10 +3083,10 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// If the maximum VF cannot produce 1 vector iteration + 1 scalar
// iteration, step down VF's to find one that can. The result should
// also eliminate any loops.
- //
- // Forced interleaving is considered when seeing if OneScalarIterationRemainder
- // is produced. It may prodiced more than one vector iteration, but only one
- // scalar iteration.
+ //
+ // Forced interleaving is considered when seeing if
+ // OneScalarIterationRemainder is produced. It may prodiced more than
+ // one vector iteration, but only one scalar iteration.
for (unsigned MaxVF = MaxFactors.FixedVF.getFixedValue(); MaxVF >= 2;
MaxVF /= 2) {
if (HasOneScalarIterationRemainder(ExactTC, MaxVF)) {
>From 55505abc0a5ffc238cb4bf9cd9b9b77e903e6b86 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 6 May 2026 08:56:45 +0100
Subject: [PATCH 05/32] Use ExpectedTC
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 7 +++----
1 file changed, 3 insertions(+), 4 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index a07e21e99c46a..1cc0f4d0ed100 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3071,15 +3071,14 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
MaxFactors.ScalableVF = ElementCount::getScalable(0);
return MaxFactors;
}
- // Allow cases where the ExactTC == VF + 1. VF can be any power of
+ // Allow cases where the ExpectedTC == VF + 1. VF can be any power of
// 2 between 2 and MaxVF.
//
// This produces 1 vector iteration, and 1 scalar iteration with
// no remainder. Later passes will eliminate the loop and leave
// straight-line code as the both iteration counts are statically known.
- ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), TheLoop);
if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
- ExactTC.isFixed()) {
+ ExpectedTC->isFixed()) {
// If the maximum VF cannot produce 1 vector iteration + 1 scalar
// iteration, step down VF's to find one that can. The result should
// also eliminate any loops.
@@ -3089,7 +3088,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// one vector iteration, but only one scalar iteration.
for (unsigned MaxVF = MaxFactors.FixedVF.getFixedValue(); MaxVF >= 2;
MaxVF /= 2) {
- if (HasOneScalarIterationRemainder(ExactTC, MaxVF)) {
+ if (HasOneScalarIterationRemainder(*ExpectedTC, MaxVF)) {
LLVM_DEBUG(dbgs() << "LV: Picking VF=" << MaxVF
<< " with 1 scalar iteration remainder.\n");
MaxFactors.FixedVF = ElementCount::getFixed(MaxVF);
>From 2c52120f3329d1e787c7c1b94ca136dbc3c8382b Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 11 May 2026 10:12:06 +0100
Subject: [PATCH 06/32] Update OPT Test and readd ExactTc
---
.../Transforms/Vectorize/LoopVectorize.cpp | 7 +-
.../sve-small-trip-count-vf-plus-one.ll | 184 ++++++++++++++++--
2 files changed, 170 insertions(+), 21 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 1cc0f4d0ed100..a07e21e99c46a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3071,14 +3071,15 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
MaxFactors.ScalableVF = ElementCount::getScalable(0);
return MaxFactors;
}
- // Allow cases where the ExpectedTC == VF + 1. VF can be any power of
+ // Allow cases where the ExactTC == VF + 1. VF can be any power of
// 2 between 2 and MaxVF.
//
// This produces 1 vector iteration, and 1 scalar iteration with
// no remainder. Later passes will eliminate the loop and leave
// straight-line code as the both iteration counts are statically known.
+ ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), TheLoop);
if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
- ExpectedTC->isFixed()) {
+ ExactTC.isFixed()) {
// If the maximum VF cannot produce 1 vector iteration + 1 scalar
// iteration, step down VF's to find one that can. The result should
// also eliminate any loops.
@@ -3088,7 +3089,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// one vector iteration, but only one scalar iteration.
for (unsigned MaxVF = MaxFactors.FixedVF.getFixedValue(); MaxVF >= 2;
MaxVF /= 2) {
- if (HasOneScalarIterationRemainder(*ExpectedTC, MaxVF)) {
+ if (HasOneScalarIterationRemainder(ExactTC, MaxVF)) {
LLVM_DEBUG(dbgs() << "LV: Picking VF=" << MaxVF
<< " with 1 scalar iteration remainder.\n");
MaxFactors.FixedVF = ElementCount::getFixed(MaxVF);
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
index 711ce3bc439d2..7e93ae7f518b3 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -13,9 +13,9 @@ target triple = "aarch64-unknown-linux-gnu"
define void @tc5_vf4_vectorize(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_vectorize(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*:]]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
@@ -23,11 +23,11 @@ define void @tc5_vf4_vectorize(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[B]], align 4
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
@@ -35,7 +35,7 @@ define void @tc5_vf4_vectorize(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -55,14 +55,110 @@ exit:
ret void
}
+; TC=5, VF=4: TC == VF + 1 (5 == 4 + 1).
+; The natural fixed-width VF for i16 is 8 on AArch64, so this also checks that
+; the low-trip-count path steps down to a smaller profitable VF.
+define void @tc5_vf4_vectorize_i16(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc5_vf4_vectorize_i16(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[A]], align 2
+; CHECK-NEXT: [[TMP0:%.*]] = add <4 x i16> [[WIDE_LOAD]], splat (i16 1)
+; CHECK-NEXT: store <4 x i16> [[TMP0]], ptr [[B]], align 2
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i16, ptr [[GEP_A]], align 2
+; CHECK-NEXT: [[ADD:%.*]] = add i16 [[VAL]], 1
+; CHECK-NEXT: store i16 [[ADD]], ptr [[GEP_B]], align 2
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i16, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i16, ptr %b, i64 %iv
+ %val = load i16, ptr %gep.a, align 2
+ %add = add i16 %val, 1
+ store i16 %add, ptr %gep.b, align 2
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop
+exit:
+ ret void
+}
+
+; TC=5, VF=4: TC == VF + 1 (5 == 4 + 1).
+; The natural fixed-width VF for i8 is 16 on AArch64, so this checks that the
+; search can step down more than once before accepting VF=4.
+define void @tc5_vf4_vectorize_i8(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc5_vf4_vectorize_i8(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[A]], align 1
+; CHECK-NEXT: [[TMP0:%.*]] = add <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; CHECK-NEXT: store <4 x i8> [[TMP0]], ptr [[B]], align 1
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i8, ptr [[GEP_A]], align 1
+; CHECK-NEXT: [[ADD:%.*]] = add i8 [[VAL]], 1
+; CHECK-NEXT: store i8 [[ADD]], ptr [[GEP_B]], align 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i8, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i8, ptr %b, i64 %iv
+ %val = load i8, ptr %gep.a, align 1
+ %add = add i8 %val, 1
+ store i8 %add, ptr %gep.b, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop
+exit:
+ ret void
+}
+
; TC=5, VF=2, IC=2: TC == VF * IC + 1 (5 == 2 * 2 + 1).
; The forced interleave count should be considered when choosing VF.
define void @tc5_forced_ic2_vectorize(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_forced_ic2_vectorize(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*:]]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 2
@@ -75,11 +171,11 @@ define void @tc5_forced_ic2_vectorize(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <2 x i32> [[TMP2]], ptr [[TMP3]], align 4
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
@@ -87,7 +183,7 @@ define void @tc5_forced_ic2_vectorize(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -107,6 +203,58 @@ exit:
ret void
}
+; TC=5, VF=2, IC=2: TC == VF * IC + 1 (5 == 2 * 2 + 1).
+; The forced interleave count should be considered when choosing VF.
+define void @tc5_forced_ic2_vectorize_i64(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc5_forced_ic2_vectorize_i64(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 2
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i64>, ptr [[A]], align 4
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <2 x i64>, ptr [[TMP0]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = add nsw <2 x i64> [[WIDE_LOAD]], splat (i64 1)
+; CHECK-NEXT: [[TMP2:%.*]] = add nsw <2 x i64> [[WIDE_LOAD1]], splat (i64 1)
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i64, ptr [[B]], i64 2
+; CHECK-NEXT: store <2 x i64> [[TMP1]], ptr [[B]], align 4
+; CHECK-NEXT: store <2 x i64> [[TMP2]], ptr [[TMP3]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i64, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i64, ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i64 [[VAL]], 1
+; CHECK-NEXT: store i64 [[ADD]], ptr [[GEP_B]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i64, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i64, ptr %b, i64 %iv
+ %val = load i64, ptr %gep.a, align 4
+ %add = add nsw i64 %val, 1
+ store i64 %add, ptr %gep.b, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop, !llvm.loop !0
+exit:
+ ret void
+}
+
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
!0 = distinct !{!0, !1}
>From 5243842f1f8fb5a153296787aa0a8c394e26d672 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 11 May 2026 14:42:25 +0100
Subject: [PATCH 07/32] Check for ExactTC!=0
The checks for matching to VF+1 == TC should protect against this, but
its best to the explicit.
---
.../Transforms/Vectorize/LoopVectorize.cpp | 2 +-
.../sve-small-trip-count-vf-plus-one.ll | 53 +++++++++++++++++++
2 files changed, 54 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index a07e21e99c46a..3b89bc8012057 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3079,7 +3079,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// straight-line code as the both iteration counts are statically known.
ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), TheLoop);
if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
- ExactTC.isFixed()) {
+ ExactTC.getFixedValue() != 0) {
// If the maximum VF cannot produce 1 vector iteration + 1 scalar
// iteration, step down VF's to find one that can. The result should
// also eliminate any loops.
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
index 7e93ae7f518b3..4b80ffb894817 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -255,6 +255,59 @@ exit:
ret void
}
+; ExactTC is unknown here because the loop trip count is the runtime value %n,
+; but the guard proves the maximum trip count is 5. This should still take the
+; low-trip-count path, but it must not use the VF+1 escape because
+; getSmallConstantTripCount returns 0.
+define void @unknown_exact_tc_max5(ptr noalias %a, ptr noalias %b, i64 %n) #0 {
+; CHECK-LABEL: define void @unknown_exact_tc_max5(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[IS_ZERO:%.*]] = icmp eq i64 [[N]], 0
+; CHECK-NEXT: br i1 [[IS_ZERO]], label %[[EXIT:.*]], label %[[GUARD:.*]]
+; CHECK: [[GUARD]]:
+; CHECK-NEXT: [[TOO_LARGE:%.*]] = icmp ugt i64 [[N]], 5
+; CHECK-NEXT: br i1 [[TOO_LARGE]], label %[[EXIT]], label %[[LOOP_PREHEADER:.*]]
+; CHECK: [[LOOP_PREHEADER]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[IV_NEXT:%.*]], %[[LOOP]] ], [ 0, %[[LOOP_PREHEADER]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[VAL]], 1
+; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT_LOOPEXIT:.*]], label %[[LOOP]]
+; CHECK: [[EXIT_LOOPEXIT]]:
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ %is.zero = icmp eq i64 %n, 0
+ br i1 %is.zero, label %exit, label %guard
+
+guard:
+ %too.large = icmp ugt i64 %n, 5
+ br i1 %too.large, label %exit, label %loop
+
+loop:
+ %iv = phi i64 [ 0, %guard ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ %val = load i32, ptr %gep.a, align 4
+ %add = add nsw i32 %val, 1
+ store i32 %add, ptr %gep.b, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, %n
+ br i1 %exitcond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
!0 = distinct !{!0, !1}
>From ed83f32129f4850d8e29535fad77bbca1893d704 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Tue, 19 May 2026 11:23:52 +0100
Subject: [PATCH 08/32] Respond to review comments
---
.../Transforms/Vectorize/LoopVectorize.cpp | 2 +-
.../sve-small-trip-count-vf-plus-one.ll | 88 +++++++++++++++++--
2 files changed, 80 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 3b89bc8012057..2c98a0ee1330a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3085,7 +3085,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// also eliminate any loops.
//
// Forced interleaving is considered when seeing if
- // OneScalarIterationRemainder is produced. It may prodiced more than
+ // OneScalarIterationRemainder is produced. It may produced more than
// one vector iteration, but only one scalar iteration.
for (unsigned MaxVF = MaxFactors.FixedVF.getFixedValue(); MaxVF >= 2;
MaxVF /= 2) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
index 4b80ffb894817..d9db7d31d3b87 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -1,17 +1,16 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -S -p loop-vectorize %s | FileCheck %s
;
; Test that a loop with trip count == VF + 1 is allowed to vectorize
; on AArch64 where under getMinTripCountTailFoldingThreshold(). This
; produces the required number of vector iteration and one scalar iteration.
-;
-; RUN: opt -S -p loop-vectorize %s | FileCheck %s
target triple = "aarch64-unknown-linux-gnu"
; TC=5, VF=4: TC == MaxFixedVF + 1 (5 == 4 + 1).
; The new code path should trigger: 1 vectorized iteration + 1 scalar iteration.
-define void @tc5_vf4_vectorize(ptr noalias %a, ptr noalias %b) #0 {
-; CHECK-LABEL: define void @tc5_vf4_vectorize(
+define void @tc5_vf4_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc5_vf4_vectorize_i32(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[SCALAR_PH1:.*:]]
; CHECK-NEXT: br label %[[LOOP1:.*]]
@@ -41,6 +40,7 @@ define void @tc5_vf4_vectorize(ptr noalias %a, ptr noalias %b) #0 {
;
entry:
br label %loop
+
loop:
%iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
%gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
@@ -51,6 +51,7 @@ loop:
%iv.next = add nuw nsw i64 %iv, 1
%exitcond = icmp eq i64 %iv.next, 5
br i1 %exitcond, label %exit, label %loop
+
exit:
ret void
}
@@ -89,6 +90,7 @@ define void @tc5_vf4_vectorize_i16(ptr noalias %a, ptr noalias %b) #0 {
;
entry:
br label %loop
+
loop:
%iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
%gep.a = getelementptr inbounds i16, ptr %a, i64 %iv
@@ -99,6 +101,7 @@ loop:
%iv.next = add nuw nsw i64 %iv, 1
%exitcond = icmp eq i64 %iv.next, 5
br i1 %exitcond, label %exit, label %loop
+
exit:
ret void
}
@@ -137,6 +140,7 @@ define void @tc5_vf4_vectorize_i8(ptr noalias %a, ptr noalias %b) #0 {
;
entry:
br label %loop
+
loop:
%iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
%gep.a = getelementptr inbounds i8, ptr %a, i64 %iv
@@ -147,14 +151,15 @@ loop:
%iv.next = add nuw nsw i64 %iv, 1
%exitcond = icmp eq i64 %iv.next, 5
br i1 %exitcond, label %exit, label %loop
+
exit:
ret void
}
; TC=5, VF=2, IC=2: TC == VF * IC + 1 (5 == 2 * 2 + 1).
; The forced interleave count should be considered when choosing VF.
-define void @tc5_forced_ic2_vectorize(ptr noalias %a, ptr noalias %b) #0 {
-; CHECK-LABEL: define void @tc5_forced_ic2_vectorize(
+define void @tc5_forced_ic2_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc5_forced_ic2_vectorize_i32(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[SCALAR_PH1:.*:]]
; CHECK-NEXT: br label %[[LOOP1:.*]]
@@ -189,6 +194,7 @@ define void @tc5_forced_ic2_vectorize(ptr noalias %a, ptr noalias %b) #0 {
;
entry:
br label %loop
+
loop:
%iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
%gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
@@ -199,6 +205,7 @@ loop:
%iv.next = add nuw nsw i64 %iv, 1
%exitcond = icmp eq i64 %iv.next, 5
br i1 %exitcond, label %exit, label %loop, !llvm.loop !0
+
exit:
ret void
}
@@ -241,6 +248,7 @@ define void @tc5_forced_ic2_vectorize_i64(ptr noalias %a, ptr noalias %b) #0 {
;
entry:
br label %loop
+
loop:
%iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
%gep.a = getelementptr inbounds i64, ptr %a, i64 %iv
@@ -251,14 +259,14 @@ loop:
%iv.next = add nuw nsw i64 %iv, 1
%exitcond = icmp eq i64 %iv.next, 5
br i1 %exitcond, label %exit, label %loop, !llvm.loop !0
+
exit:
ret void
}
; ExactTC is unknown here because the loop trip count is the runtime value %n,
-; but the guard proves the maximum trip count is 5. This should still take the
-; low-trip-count path, but it must not use the VF+1 escape because
-; getSmallConstantTripCount returns 0.
+; but the guard proves the maximum trip count is 5. it must not use the VF+1
+; escape because getSmallConstantTripCount returns 0.
define void @unknown_exact_tc_max5(ptr noalias %a, ptr noalias %b, i64 %n) #0 {
; CHECK-LABEL: define void @unknown_exact_tc_max5(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
@@ -308,7 +316,69 @@ exit:
ret void
}
+; TC=5, VF=4, UserIC=4
+; The user interleave count should be ignored because the dependence distance
+; makes the loop unsafe for interleaving > 1. Vectorization should still pick
+; VF=4 and produce 1 vector iteration plus 1 scalar iteration.
+define void @tc5_vf4_unsafe_useric_distance4_i32(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc5_vf4_unsafe_useric_distance4_i32(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 4
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[TMP0]], i64 -4
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 4
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = add nsw <4 x i32> [[WIDE_LOAD1]], [[WIDE_LOAD]]
+; CHECK-NEXT: store <4 x i32> [[TMP3]], ptr [[TMP0]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 8, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[B_DST:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[B_SRC:%.*]] = getelementptr inbounds i32, ptr [[B_DST]], i64 -4
+; CHECK-NEXT: [[DEP:%.*]] = load i32, ptr [[B_SRC]], align 4
+; CHECK-NEXT: [[A_SRC:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[A_SRC]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[VAL]], [[DEP]]
+; CHECK-NEXT: store i32 [[ADD]], ptr [[B_DST]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 9
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 4, %entry ], [ %iv.next, %loop ]
+ %b.dst = getelementptr inbounds i32, ptr %b, i64 %iv
+ %b.src = getelementptr inbounds i32, ptr %b.dst, i64 -4
+ %dep = load i32, ptr %b.src, align 4
+ %a.src = getelementptr inbounds i32, ptr %a, i64 %iv
+ %val = load i32, ptr %a.src, align 4
+ %add = add nsw i32 %val, %dep
+ store i32 %add, ptr %b.dst, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 9
+ br i1 %exitcond, label %exit, label %loop, !llvm.loop !2
+
+exit:
+ ret void
+}
+
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
!0 = distinct !{!0, !1}
!1 = !{!"llvm.loop.interleave.count", i32 2}
+!2 = distinct !{!2, !3, !4}
+!3 = !{!"llvm.loop.interleave.count", i32 4}
+!4 = !{!"llvm.loop.vectorize.width", i32 4}
>From 6d14d2c33cbd5e18670b3a263fd5136d950ed289 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 20 May 2026 10:23:41 +0000
Subject: [PATCH 09/32] Reuse EffectiveIC in NoScalarEpilogueNeeded
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 10 +++++-----
1 file changed, 5 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 2c98a0ee1330a..a3346afdac4d7 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3018,13 +3018,13 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
MaxPowerOf2RuntimeVF = std::nullopt; // Stick with tail-folding for now.
}
- auto NoScalarEpilogueNeeded = [this, &UserIC](unsigned MaxVF) {
+ auto NoScalarEpilogueNeeded = [this](unsigned MaxVF, unsigned EffectiveIC) {
// Return false if the loop is neither a single-latch-exit loop nor an
// early-exit loop as tail-folding is not supported in that case.
if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
!Legal->hasUncountableEarlyExit())
return false;
- unsigned MaxVFtimesIC = UserIC ? MaxVF * UserIC : MaxVF;
+ unsigned MaxVFtimesIC = MaxVF * EffectiveIC;
ScalarEvolution *SE = PSE.getSE();
// Calling getSymbolicMaxBackedgeTakenCount enables support for loops
// with uncountable exits. For countable loops, the symbolic maximum must
@@ -3041,10 +3041,11 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
return Rem->isZero();
};
+ unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
if (MaxPowerOf2RuntimeVF > 0u) {
assert((UserVF.isNonZero() || isPowerOf2_32(*MaxPowerOf2RuntimeVF)) &&
"MaxFixedVF must be a power of 2");
- if (NoScalarEpilogueNeeded(*MaxPowerOf2RuntimeVF)) {
+ if (NoScalarEpilogueNeeded(*MaxPowerOf2RuntimeVF, EffectiveIC)) {
// Accept MaxFixedVF if we do not have a tail.
LLVM_DEBUG(dbgs() << "LV: No tail will remain for any chosen VF.\n");
return MaxFactors;
@@ -3052,7 +3053,6 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
}
auto ExpectedTC = getSmallBestKnownTC(PSE, TheLoop);
- unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
auto HasOneScalarIterationRemainder =
[EffectiveIC](ElementCount &ExactTC, unsigned int MaxVF) -> bool {
return ExactTC.getFixedValue() == 1 + (MaxVF * EffectiveIC);
@@ -3065,7 +3065,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// the trip count but the scalable factor does not, use the fixed-width
// factor in preference to allow the generation of a non-predicated loop.
if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
- NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue())) {
+ NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue(), EffectiveIC)) {
LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width so that no tail will "
"remain for any chosen VF.\n");
MaxFactors.ScalableVF = ElementCount::getScalable(0);
>From ec7cc1d159f17f2f76be363031230c6e454ad2b5 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 20 May 2026 10:51:36 +0000
Subject: [PATCH 10/32] format
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 3 ++-
1 file changed, 2 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index a3346afdac4d7..3cfcce92c0e42 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3065,7 +3065,8 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// the trip count but the scalable factor does not, use the fixed-width
// factor in preference to allow the generation of a non-predicated loop.
if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
- NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue(), EffectiveIC)) {
+ NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue(),
+ EffectiveIC)) {
LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width so that no tail will "
"remain for any chosen VF.\n");
MaxFactors.ScalableVF = ElementCount::getScalable(0);
>From ae098e17b78c72737445aef19a5fe7809c6b4bee Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 28 May 2026 11:51:40 +0100
Subject: [PATCH 11/32] Address nit comments
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 8 +++-----
.../AArch64/sve-small-trip-count-vf-plus-one.ll | 10 +++++-----
2 files changed, 8 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 3cfcce92c0e42..f7346dc47d368 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3084,15 +3084,13 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// If the maximum VF cannot produce 1 vector iteration + 1 scalar
// iteration, step down VF's to find one that can. The result should
// also eliminate any loops.
- //
- // Forced interleaving is considered when seeing if
- // OneScalarIterationRemainder is produced. It may produced more than
- // one vector iteration, but only one scalar iteration.
for (unsigned MaxVF = MaxFactors.FixedVF.getFixedValue(); MaxVF >= 2;
MaxVF /= 2) {
+ // OneScalarIterationRemainder takes account of any forced
+ // interleaving.
if (HasOneScalarIterationRemainder(ExactTC, MaxVF)) {
LLVM_DEBUG(dbgs() << "LV: Picking VF=" << MaxVF
- << " with 1 scalar iteration remainder.\n");
+ << " with 1 scalar iteration remaining.\n");
MaxFactors.FixedVF = ElementCount::getFixed(MaxVF);
MaxFactors.ScalableVF = ElementCount::getScalable(0);
return MaxFactors;
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
index d9db7d31d3b87..5c08f97a55aa2 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -57,7 +57,7 @@ exit:
}
; TC=5, VF=4: TC == VF + 1 (5 == 4 + 1).
-; The natural fixed-width VF for i16 is 8 on AArch64, so this also checks that
+; VF=8 is a natural fixed-width for i16 types on AArch64, so this also checks that
; the low-trip-count path steps down to a smaller profitable VF.
define void @tc5_vf4_vectorize_i16(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_vectorize_i16(
@@ -107,7 +107,7 @@ exit:
}
; TC=5, VF=4: TC == VF + 1 (5 == 4 + 1).
-; The natural fixed-width VF for i8 is 16 on AArch64, so this checks that the
+; VF=16 is a natural fixed-width for i8 types on AArch64, so this checks that the
; search can step down more than once before accepting VF=4.
define void @tc5_vf4_vectorize_i8(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_vectorize_i8(
@@ -264,9 +264,9 @@ exit:
ret void
}
-; ExactTC is unknown here because the loop trip count is the runtime value %n,
-; but the guard proves the maximum trip count is 5. it must not use the VF+1
-; escape because getSmallConstantTripCount returns 0.
+; In this case the vectoriser shouldn't optimise for a single vector iteration
+; + single scalar iteration, because there is no guarantee we will enter the vector
+; loop.
define void @unknown_exact_tc_max5(ptr noalias %a, ptr noalias %b, i64 %n) #0 {
; CHECK-LABEL: define void @unknown_exact_tc_max5(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
>From 67511570895d7736b4749f7b1b4e35e04582ee06 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Fri, 29 May 2026 09:54:07 +0100
Subject: [PATCH 12/32] Add cost modelling for where scalar loops are more
profitable
---
.../Vectorize/LoopVectorizationPlanner.cpp | 42 +++---
.../Vectorize/LoopVectorizationPlanner.h | 5 +
.../Transforms/Vectorize/LoopVectorize.cpp | 77 ++++++++++
.../sve-small-trip-count-vf-plus-one-cost.ll | 132 ++++++++++++++++++
4 files changed, 237 insertions(+), 19 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index cbe2f63f96005..33802c501033b 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -768,25 +768,8 @@ bool LoopVectorizationPlanner::isMoreProfitable(const VectorizationFactor &A,
if (!MaxTripCount)
return LowerCostWithoutTC;
- auto GetCostForTC = [MaxTripCount, HasTail](unsigned VF,
- InstructionCost VectorCost,
- InstructionCost ScalarCost) {
- // If the trip count is a known (possibly small) constant, the trip count
- // will be rounded up to an integer number of iterations under
- // FoldTailByMasking. The total cost in that case will be
- // VecCost*ceil(TripCount/VF). When not folding the tail, the total
- // cost will be VecCost*floor(TC/VF) + ScalarCost*(TC%VF). There will be
- // some extra overheads, but for the purpose of comparing the costs of
- // different VFs we can use this to compare the total loop-body cost
- // expected after vectorization.
- if (HasTail)
- return VectorCost * (MaxTripCount / VF) +
- ScalarCost * (MaxTripCount % VF);
- return VectorCost * divideCeil(MaxTripCount, VF);
- };
-
- auto RTCostA = GetCostForTC(EstimatedWidthA, CostA, A.ScalarCost);
- auto RTCostB = GetCostForTC(EstimatedWidthB, CostB, B.ScalarCost);
+ auto RTCostA = getCostForKnownTripCount(A, MaxTripCount, HasTail);
+ auto RTCostB = getCostForKnownTripCount(B, MaxTripCount, HasTail);
bool LowerCostWithTC = CmpFn(RTCostA, RTCostB);
LLVM_DEBUG(if (LowerCostWithTC != LowerCostWithoutTC) {
dbgs() << "LV: VF " << (LowerCostWithTC ? A.Width : B.Width)
@@ -808,6 +791,27 @@ bool LoopVectorizationPlanner::isMoreProfitable(const VectorizationFactor &A,
IsEpilogue);
}
+InstructionCost LoopVectorizationPlanner::getCostForKnownTripCount(
+ const VectorizationFactor &VF, unsigned TripCount, bool HasTail) const {
+ unsigned EstimatedWidth = VF.Width.getKnownMinValue();
+ if (std::optional<unsigned> VScale = Config.getVScaleForTuning())
+ if (VF.Width.isScalable())
+ EstimatedWidth *= *VScale;
+
+ // If the trip count is a known (possibly small) constant, the trip count
+ // will be rounded up to an integer number of iterations under
+ // FoldTailByMasking. The total cost in that case will be
+ // VecCost*ceil(TripCount/VF). When not folding the tail, the total
+ // cost will be VecCost*floor(TC/VF) + ScalarCost*(TC%VF). There will be
+ // some extra overheads, but for the purpose of comparing the costs of
+ // different VFs we can use this to compare the total loop-body cost
+ // expected after vectorization.
+ if (HasTail)
+ return VF.Cost * (TripCount / EstimatedWidth) +
+ VF.ScalarCost * (TripCount % EstimatedWidth);
+ return VF.Cost * divideCeil(TripCount, EstimatedWidth);
+}
+
// TODO: we could return a pair of values that specify the max VF and
// min VF, to be used in `buildVPlans(MinVF, MaxVF)` instead of
// `buildVPlans(VF, VF)`. We cannot do it because VPLAN at the moment
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index cca1d4dbc5720..0fd3913372cff 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -1046,6 +1046,11 @@ class LoopVectorizationPlanner {
const unsigned MaxTripCount, bool HasTail,
bool IsEpilogue = false) const;
+ /// Returns the estimated loop-body cost for \p VF and a known trip count.
+ InstructionCost getCostForKnownTripCount(const VectorizationFactor &VF,
+ unsigned TripCount,
+ bool HasTail) const;
+
/// Determines if we have the infrastructure to vectorize the loop and its
/// epilogue, assuming the main loop is vectorized by \p MainPlan.
bool isCandidateForEpilogueVectorization(VPlan &MainPlan) const;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index f7346dc47d368..02eb485ee250e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5848,6 +5848,46 @@ LoopVectorizationPlanner::computeBestVF() {
assert((FirstPlan.hasScalarVFOnly() || hasPlanWithVF(UserVF) ||
FirstPlan.isOuterLoop()) &&
"must have a single scalar VF, UserVF or an outer loop");
+ bool ForceVectorization =
+ Hints.getForce() == LoopVectorizeHints::FK_Enabled;
+ if (!FirstPlan.hasScalarVFOnly() && !FirstPlan.isOuterLoop() &&
+ hasPlanWithVF(UserVF) && UserVF.isVector() && !ForceVectorization) {
+ ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), OrigLoop);
+ if (FirstPlan.hasScalarTail() && ExactTC.isFixed() && UserVF.isFixed()) {
+ unsigned TC = ExactTC.getFixedValue();
+ unsigned EstimatedWidth =
+ estimateElementCount(UserVF, Config.getVScaleForTuning());
+ if (TC != 0 && TC <= TTI.getMinTripCountTailFoldingThreshold() &&
+ TC == EstimatedWidth + 1) {
+ ElementCount ScalarVF = ElementCount::getFixed(1);
+ InstructionCost ScalarCost = CM.expectedCost(ScalarVF);
+ LLVM_DEBUG(dbgs()
+ << "LV: Scalar loop costs: " << ScalarCost << ".\n");
+
+ InstructionCost Cost = cost(FirstPlan, UserVF, /*RU=*/nullptr);
+ VectorizationFactor UserFactor(UserVF, Cost, ScalarCost);
+
+ InstructionCost VectorCost =
+ getCostForKnownTripCount(UserFactor, TC, /*HasTail=*/true);
+ VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
+ InstructionCost ScalarCostForTC =
+ getCostForKnownTripCount(ScalarFactor, TC, /*HasTail=*/false);
+ // Be conservative for the one-scalar-tail shape. It introduces
+ // extra control flow and a scalar epilogue for a single element, so
+ // require the vectorized form to save at least one scalar iteration.
+ InstructionCost AdjustedVectorCost = VectorCost + ScalarCost;
+ if (VectorCost.isValid() && ScalarCostForTC.isValid() &&
+ AdjustedVectorCost >= ScalarCostForTC) {
+ LLVM_DEBUG(dbgs()
+ << "LV: Rejecting VF " << UserVF
+ << " for one-scalar-tail low trip count: vector cost "
+ << AdjustedVectorCost << " >= scalar cost "
+ << ScalarCostForTC << ".\n");
+ return {ScalarFactor, &FirstPlan};
+ }
+ }
+ }
+ }
return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
}
@@ -5890,6 +5930,40 @@ LoopVectorizationPlanner::computeBestVF() {
}
VPlan *PlanForBestVF = &FirstPlan;
+ ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), OrigLoop);
+ auto IsUnprofitableOneScalarTail =
+ [&](const VectorizationFactor &CurrentFactor, bool HasTail) {
+ if (ForceVectorization || !HasTail || !ExactTC.isFixed() ||
+ CurrentFactor.Width.isScalable())
+ return false;
+
+ unsigned TC = ExactTC.getFixedValue();
+ if (TC == 0 || TC > TTI.getMinTripCountTailFoldingThreshold())
+ return false;
+
+ unsigned EstimatedWidth = estimateElementCount(
+ CurrentFactor.Width, Config.getVScaleForTuning());
+ if (TC % EstimatedWidth != 1)
+ return false;
+
+ InstructionCost VectorCost =
+ getCostForKnownTripCount(CurrentFactor, TC, HasTail);
+ InstructionCost ScalarCostForTC =
+ getCostForKnownTripCount(ScalarFactor, TC, /*HasTail=*/false);
+ // Be conservative for the one-scalar-tail shape. It introduces extra
+ // control flow and a scalar epilogue for a single element, so require
+ // the vectorized form to save at least one scalar iteration.
+ InstructionCost AdjustedVectorCost = VectorCost + ScalarCost;
+ if (!VectorCost.isValid() || !ScalarCostForTC.isValid() ||
+ AdjustedVectorCost < ScalarCostForTC)
+ return false;
+
+ LLVM_DEBUG(dbgs() << "LV: Rejecting VF " << CurrentFactor.Width
+ << " for one-scalar-tail low trip count: vector cost "
+ << AdjustedVectorCost << " >= scalar cost "
+ << ScalarCostForTC << ".\n");
+ return true;
+ };
for (auto &P : VPlans) {
ArrayRef<ElementCount> VFs(P->vectorFactors().begin(),
@@ -5926,6 +6000,9 @@ LoopVectorizationPlanner::computeBestVF() {
cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
+ if (IsUnprofitableOneScalarTail(CurrentFactor, P->hasScalarTail()))
+ continue;
+
if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
BestFactor = CurrentFactor;
PlanForBestVF = P.get();
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
new file mode 100644
index 0000000000000..82cbb431ed76f
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
@@ -0,0 +1,132 @@
+; REQUIRES: asserts
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve -S %s | FileCheck %s --check-prefix=IR
+; RUN: opt -passes=loop-vectorize -mtriple=aarch64-linux-gnu -mattr=+sve -debug-only=loop-vectorize -disable-output %s 2>&1 | FileCheck %s --check-prefix=DBG
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @tc3_udiv_i8_reject(ptr noalias %a, ptr noalias %b,
+ ptr noalias %c) #0 {
+; IR-LABEL: define void @tc3_udiv_i8_reject(
+; IR-NOT: vector.body
+; IR-LABEL: define void @tc3_udiv_i8_user_vf2(
+; IR-NOT: vector.body
+; IR-LABEL: define void @tc3_smin_i8_reject(
+; IR-NOT: vector.body
+; IR-LABEL: define void @tc3_udiv_i8_forced(
+; IR: vector.body:
+;
+; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_reject'
+; DBG: LV: Picking VF=2 with 1 scalar iteration remaining.
+; DBG: LV: Scalar loop costs: 9.
+; DBG: Cost for VF 2: 19
+; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 37 >= scalar cost 27.
+; DBG: LV: Selecting VF: 1.
+; DBG: LV: Vectorization is possible but not beneficial.
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %pa = getelementptr inbounds i8, ptr %a, i64 %iv
+ %pb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %pc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %va = load i8, ptr %pa, align 1
+ %vb = load i8, ptr %pb, align 1
+ %div = udiv i8 %va, %vb
+ store i8 %div, ptr %pc, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @tc3_udiv_i8_user_vf2(ptr noalias %a, ptr noalias %b,
+ ptr noalias %c) #0 {
+; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_user_vf2'
+; DBG: LV: Using user VF 2.
+; DBG: LV: Scalar loop costs: 9.
+; DBG: Cost for VF 2: 19
+; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 37 >= scalar cost 27.
+; DBG: LV: Vectorization is possible but not beneficial.
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %pa = getelementptr inbounds i8, ptr %a, i64 %iv
+ %pb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %pc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %va = load i8, ptr %pa, align 1
+ %vb = load i8, ptr %pb, align 1
+ %div = udiv i8 %va, %vb
+ store i8 %div, ptr %pc, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop, !llvm.loop !2
+
+exit:
+ ret void
+}
+
+define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
+; DBG-LABEL: LV: Checking a loop in 'tc3_smin_i8_reject'
+; DBG: LV: Picking VF=2 with 1 scalar iteration remaining.
+; DBG: LV: Scalar loop costs: 10.
+; DBG: Cost for VF 2: 15
+; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 35 >= scalar cost 30.
+; DBG: LV: Selecting VF: 1.
+; DBG: LV: Vectorization is possible but not beneficial.
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %arrayidx = getelementptr inbounds i8, ptr %b, i64 %iv
+ %0 = load i8, ptr %arrayidx, align 1
+ %arrayidx2 = getelementptr inbounds i8, ptr %a, i64 %iv
+ %1 = load i8, ptr %arrayidx2, align 1
+ %min = tail call i8 @llvm.smin.i8(i8 %0, i8 %1)
+ store i8 %min, ptr %arrayidx, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @tc3_udiv_i8_forced(ptr noalias %a, ptr noalias %b,
+ ptr noalias %c) #0 {
+; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_forced'
+; DBG-NOT: Rejecting VF 2
+; DBG: LV: Selecting VF: 2.
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %pa = getelementptr inbounds i8, ptr %a, i64 %iv
+ %pb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %pc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %va = load i8, ptr %pa, align 1
+ %vb = load i8, ptr %pb, align 1
+ %div = udiv i8 %va, %vb
+ store i8 %div, ptr %pc, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop, !llvm.loop !0
+
+exit:
+ ret void
+}
+
+declare i8 @llvm.smin.i8(i8, i8)
+
+attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.vectorize.enable", i1 true}
+!2 = distinct !{!2, !3}
+!3 = !{!"llvm.loop.vectorize.width", i32 2}
>From f905b06daf2f986e7680acdcd2f37ca3c634b45b Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Fri, 29 May 2026 14:06:50 +0100
Subject: [PATCH 13/32] Remove for loop for calculating best VF
---
.../Transforms/Vectorize/LoopVectorize.cpp | 25 +++++++------------
1 file changed, 9 insertions(+), 16 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 02eb485ee250e..b483dd112f10d 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3053,10 +3053,6 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
}
auto ExpectedTC = getSmallBestKnownTC(PSE, TheLoop);
- auto HasOneScalarIterationRemainder =
- [EffectiveIC](ElementCount &ExactTC, unsigned int MaxVF) -> bool {
- return ExactTC.getFixedValue() == 1 + (MaxVF * EffectiveIC);
- };
if (ExpectedTC && ExpectedTC->isFixed() &&
ExpectedTC->getFixedValue() <=
TTI.getMinTripCountTailFoldingThreshold()) {
@@ -3080,18 +3076,15 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// straight-line code as the both iteration counts are statically known.
ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), TheLoop);
if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
- ExactTC.getFixedValue() != 0) {
- // If the maximum VF cannot produce 1 vector iteration + 1 scalar
- // iteration, step down VF's to find one that can. The result should
- // also eliminate any loops.
- for (unsigned MaxVF = MaxFactors.FixedVF.getFixedValue(); MaxVF >= 2;
- MaxVF /= 2) {
- // OneScalarIterationRemainder takes account of any forced
- // interleaving.
- if (HasOneScalarIterationRemainder(ExactTC, MaxVF)) {
- LLVM_DEBUG(dbgs() << "LV: Picking VF=" << MaxVF
+ ExactTC.getFixedValue() > 1) {
+ unsigned TC = ExactTC.getFixedValue();
+ unsigned MaxFixedVF = MaxFactors.FixedVF.getFixedValue();
+ if ((TC - 1) % EffectiveIC == 0) {
+ unsigned VF = (TC - 1) / EffectiveIC;
+ if (VF >= 2 && VF <= MaxFixedVF && isPowerOf2_32(VF)) {
+ LLVM_DEBUG(dbgs() << "LV: Picking VF=" << VF
<< " with 1 scalar iteration remaining.\n");
- MaxFactors.FixedVF = ElementCount::getFixed(MaxVF);
+ MaxFactors.FixedVF = ElementCount::getFixed(VF);
MaxFactors.ScalableVF = ElementCount::getScalable(0);
return MaxFactors;
}
@@ -5943,7 +5936,7 @@ LoopVectorizationPlanner::computeBestVF() {
unsigned EstimatedWidth = estimateElementCount(
CurrentFactor.Width, Config.getVScaleForTuning());
- if (TC % EstimatedWidth != 1)
+ if (TC != EstimatedWidth + 1)
return false;
InstructionCost VectorCost =
>From 7cbee03efecae1778a58a616fe1e3d1df1fd035d Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Tue, 23 Jun 2026 09:30:56 +0100
Subject: [PATCH 14/32] Refactor IsUnprofitableOneScalarTail Lambda Function
---
.../Transforms/Vectorize/LoopVectorize.cpp | 92 +++++++++----------
1 file changed, 42 insertions(+), 50 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index b483dd112f10d..ad80b01d209c7 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5833,6 +5833,42 @@ LoopVectorizationPlanner::computeBestVF() {
return {VectorizationFactor::Disabled(), nullptr};
// If there is a single VPlan with a single VF, return it directly.
VPlan &FirstPlan = *VPlans[0];
+ auto IsUnprofitableOneScalarTail =
+ [&](const VectorizationFactor &CurrentFactor, bool HasTail,
+ bool ForceVectorization, const ElementCount &ExactTC,
+ const VectorizationFactor &ScalarFactor,
+ const InstructionCost &ScalarCost) {
+ if (ForceVectorization || !HasTail || !ExactTC.isFixed() ||
+ CurrentFactor.Width.isScalable())
+ return false;
+
+ unsigned TC = ExactTC.getFixedValue();
+ if (TC == 0 || TC > TTI.getMinTripCountTailFoldingThreshold())
+ return false;
+
+ unsigned EstimatedWidth = estimateElementCount(
+ CurrentFactor.Width, Config.getVScaleForTuning());
+ if (TC != EstimatedWidth + 1)
+ return false;
+
+ InstructionCost VectorCost =
+ getCostForKnownTripCount(CurrentFactor, TC, HasTail);
+ InstructionCost ScalarCostForTC =
+ getCostForKnownTripCount(ScalarFactor, TC, /*HasTail=*/false);
+ // Be conservative for the one-scalar-tail shape. It introduces extra
+ // control flow and a scalar epilogue for a single element, so require
+ // the vectorized form to save at least one scalar iteration.
+ InstructionCost AdjustedVectorCost = VectorCost + ScalarCost;
+ if (!AdjustedVectorCost.isValid() ||
+ AdjustedVectorCost < ScalarCostForTC)
+ return false;
+
+ LLVM_DEBUG(dbgs() << "LV: Rejecting VF " << CurrentFactor.Width
+ << " for one-scalar-tail low trip count: vector cost "
+ << AdjustedVectorCost << " >= scalar cost "
+ << ScalarCostForTC << ".\n");
+ return true;
+ };
ElementCount UserVF = Config.getHints().getWidth();
if (VPlans.size() == 1) {
@@ -5859,23 +5895,10 @@ LoopVectorizationPlanner::computeBestVF() {
InstructionCost Cost = cost(FirstPlan, UserVF, /*RU=*/nullptr);
VectorizationFactor UserFactor(UserVF, Cost, ScalarCost);
-
- InstructionCost VectorCost =
- getCostForKnownTripCount(UserFactor, TC, /*HasTail=*/true);
VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
- InstructionCost ScalarCostForTC =
- getCostForKnownTripCount(ScalarFactor, TC, /*HasTail=*/false);
- // Be conservative for the one-scalar-tail shape. It introduces
- // extra control flow and a scalar epilogue for a single element, so
- // require the vectorized form to save at least one scalar iteration.
- InstructionCost AdjustedVectorCost = VectorCost + ScalarCost;
- if (VectorCost.isValid() && ScalarCostForTC.isValid() &&
- AdjustedVectorCost >= ScalarCostForTC) {
- LLVM_DEBUG(dbgs()
- << "LV: Rejecting VF " << UserVF
- << " for one-scalar-tail low trip count: vector cost "
- << AdjustedVectorCost << " >= scalar cost "
- << ScalarCostForTC << ".\n");
+ if (IsUnprofitableOneScalarTail(UserFactor, FirstPlan.hasScalarTail(),
+ ForceVectorization, ExactTC,
+ ScalarFactor, ScalarCost)) {
return {ScalarFactor, &FirstPlan};
}
}
@@ -5924,39 +5947,6 @@ LoopVectorizationPlanner::computeBestVF() {
VPlan *PlanForBestVF = &FirstPlan;
ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), OrigLoop);
- auto IsUnprofitableOneScalarTail =
- [&](const VectorizationFactor &CurrentFactor, bool HasTail) {
- if (ForceVectorization || !HasTail || !ExactTC.isFixed() ||
- CurrentFactor.Width.isScalable())
- return false;
-
- unsigned TC = ExactTC.getFixedValue();
- if (TC == 0 || TC > TTI.getMinTripCountTailFoldingThreshold())
- return false;
-
- unsigned EstimatedWidth = estimateElementCount(
- CurrentFactor.Width, Config.getVScaleForTuning());
- if (TC != EstimatedWidth + 1)
- return false;
-
- InstructionCost VectorCost =
- getCostForKnownTripCount(CurrentFactor, TC, HasTail);
- InstructionCost ScalarCostForTC =
- getCostForKnownTripCount(ScalarFactor, TC, /*HasTail=*/false);
- // Be conservative for the one-scalar-tail shape. It introduces extra
- // control flow and a scalar epilogue for a single element, so require
- // the vectorized form to save at least one scalar iteration.
- InstructionCost AdjustedVectorCost = VectorCost + ScalarCost;
- if (!VectorCost.isValid() || !ScalarCostForTC.isValid() ||
- AdjustedVectorCost < ScalarCostForTC)
- return false;
-
- LLVM_DEBUG(dbgs() << "LV: Rejecting VF " << CurrentFactor.Width
- << " for one-scalar-tail low trip count: vector cost "
- << AdjustedVectorCost << " >= scalar cost "
- << ScalarCostForTC << ".\n");
- return true;
- };
for (auto &P : VPlans) {
ArrayRef<ElementCount> VFs(P->vectorFactors().begin(),
@@ -5993,7 +5983,9 @@ LoopVectorizationPlanner::computeBestVF() {
cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
- if (IsUnprofitableOneScalarTail(CurrentFactor, P->hasScalarTail()))
+ if (IsUnprofitableOneScalarTail(CurrentFactor, P->hasScalarTail(),
+ ForceVectorization, ExactTC, ScalarFactor,
+ ScalarCost))
continue;
if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
>From 1d149ae440cefbb12e2bed43c31b41e39141c0a6 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 2 Jul 2026 15:54:45 +0100
Subject: [PATCH 15/32] Add consideration of IC and test
---
.../Transforms/Vectorize/LoopVectorize.cpp | 11 +++--
.../sve-small-trip-count-vf-plus-one-cost.ll | 46 ++++++++++++++++---
2 files changed, 45 insertions(+), 12 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ad80b01d209c7..7a975351cb12e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5837,7 +5837,7 @@ LoopVectorizationPlanner::computeBestVF() {
[&](const VectorizationFactor &CurrentFactor, bool HasTail,
bool ForceVectorization, const ElementCount &ExactTC,
const VectorizationFactor &ScalarFactor,
- const InstructionCost &ScalarCost) {
+ const InstructionCost &ScalarCost, unsigned int UserIC) {
if (ForceVectorization || !HasTail || !ExactTC.isFixed() ||
CurrentFactor.Width.isScalable())
return false;
@@ -5848,7 +5848,7 @@ LoopVectorizationPlanner::computeBestVF() {
unsigned EstimatedWidth = estimateElementCount(
CurrentFactor.Width, Config.getVScaleForTuning());
- if (TC != EstimatedWidth + 1)
+ if (TC != (EstimatedWidth * UserIC) + 1)
return false;
InstructionCost VectorCost =
@@ -5871,6 +5871,7 @@ LoopVectorizationPlanner::computeBestVF() {
};
ElementCount UserVF = Config.getHints().getWidth();
+ unsigned int UserIC = Config.getHints().getInterleave() != 0 ? Config.getHints().getInterleave() : 1;
if (VPlans.size() == 1) {
// For outer loops, the plan has a single vector VF determined by the
// heuristic.
@@ -5878,7 +5879,7 @@ LoopVectorizationPlanner::computeBestVF() {
FirstPlan.isOuterLoop()) &&
"must have a single scalar VF, UserVF or an outer loop");
bool ForceVectorization =
- Hints.getForce() == LoopVectorizeHints::FK_Enabled;
+ Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled;
if (!FirstPlan.hasScalarVFOnly() && !FirstPlan.isOuterLoop() &&
hasPlanWithVF(UserVF) && UserVF.isVector() && !ForceVectorization) {
ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), OrigLoop);
@@ -5898,7 +5899,7 @@ LoopVectorizationPlanner::computeBestVF() {
VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
if (IsUnprofitableOneScalarTail(UserFactor, FirstPlan.hasScalarTail(),
ForceVectorization, ExactTC,
- ScalarFactor, ScalarCost)) {
+ ScalarFactor, ScalarCost, UserIC)) {
return {ScalarFactor, &FirstPlan};
}
}
@@ -5985,7 +5986,7 @@ LoopVectorizationPlanner::computeBestVF() {
if (IsUnprofitableOneScalarTail(CurrentFactor, P->hasScalarTail(),
ForceVectorization, ExactTC, ScalarFactor,
- ScalarCost))
+ ScalarCost, UserIC))
continue;
if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
index 82cbb431ed76f..70eb5b07c25aa 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
@@ -7,13 +7,7 @@ target triple = "aarch64-unknown-linux-gnu"
define void @tc3_udiv_i8_reject(ptr noalias %a, ptr noalias %b,
ptr noalias %c) #0 {
; IR-LABEL: define void @tc3_udiv_i8_reject(
-; IR-NOT: vector.body
-; IR-LABEL: define void @tc3_udiv_i8_user_vf2(
-; IR-NOT: vector.body
-; IR-LABEL: define void @tc3_smin_i8_reject(
-; IR-NOT: vector.body
-; IR-LABEL: define void @tc3_udiv_i8_forced(
-; IR: vector.body:
+; IR-NOT: vector.body:
;
; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_reject'
; DBG: LV: Picking VF=2 with 1 scalar iteration remaining.
@@ -44,6 +38,9 @@ exit:
define void @tc3_udiv_i8_user_vf2(ptr noalias %a, ptr noalias %b,
ptr noalias %c) #0 {
+; IR-LABEL: define void @tc3_udiv_i8_user_vf2(
+; IR-NOT: vector.body
+
; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_user_vf2'
; DBG: LV: Using user VF 2.
; DBG: LV: Scalar loop costs: 9.
@@ -71,6 +68,9 @@ exit:
}
define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
+; IR-LABEL: define void @tc3_smin_i8_reject(
+; IR-NOT: vector.body
+
; DBG-LABEL: LV: Checking a loop in 'tc3_smin_i8_reject'
; DBG: LV: Picking VF=2 with 1 scalar iteration remaining.
; DBG: LV: Scalar loop costs: 10.
@@ -99,6 +99,9 @@ exit:
define void @tc3_udiv_i8_forced(ptr noalias %a, ptr noalias %b,
ptr noalias %c) #0 {
+; IR-LABEL: define void @tc3_udiv_i8_forced(
+; IR: vector.body
+
; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_forced'
; DBG-NOT: Rejecting VF 2
; DBG: LV: Selecting VF: 2.
@@ -122,6 +125,33 @@ exit:
ret void
}
+define void @tc5_udiv_i8_reject_ic2(ptr noalias %a, ptr noalias %b, ptr noalias %c) #0{
+; IR-LABEL: define void @tc5_udiv_i8_reject_ic2(
+; IR: vector.body
+
+; DBG-LABEL: LV: Checking a loop in 'tc5_udiv_i8_reject_ic2'
+; DBG-NOT: LV: Selecting VF: 2.
+; DBG Rejecting VF 2
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %pa = getelementptr inbounds i8, ptr %a, i64 %iv
+ %pb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %pc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %va = load i8, ptr %pa, align 1
+ %vb = load i8, ptr %pb, align 1
+ %div = udiv i8 %va, %vb
+ store i8 %div, ptr %pc, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop, !llvm.loop !4
+
+exit:
+ ret void
+}
+
declare i8 @llvm.smin.i8(i8, i8)
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
@@ -130,3 +160,5 @@ attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
!1 = !{!"llvm.loop.vectorize.enable", i1 true}
!2 = distinct !{!2, !3}
!3 = !{!"llvm.loop.vectorize.width", i32 2}
+!4 = distinct !{!4, !5}
+!5 = !{!"llvm.loop.interleave.count", i32 2}
>From b2ac4c7b636ed0e2b30e764fe456259a8d4a5960 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 16 Jul 2026 15:25:20 +0100
Subject: [PATCH 16/32] Respond to review comments
---
.../Vectorize/LoopVectorizationPlanner.h | 7 ++
.../Transforms/Vectorize/LoopVectorize.cpp | 81 ++++++++++---------
.../sve-small-trip-count-vf-plus-one-cost.ll | 75 +++++++++--------
.../sve-small-trip-count-vf-plus-one.ll | 81 +++++++++++++++++++
4 files changed, 166 insertions(+), 78 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 0fd3913372cff..26cee9e7768ce 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -919,6 +919,13 @@ class LoopVectorizationPlanner {
/// for each VF.
VPlan &getPlanFor(ElementCount VF) const;
+ /// Examines if it is unprofitable to Vectorize a small loop in a way that leaves a
+ /// Vector iteration, followed by a single iteration scalar tail. For some uses cases,
+ /// it is better to leave the original Scalar loop in place.
+ bool isUnprofitableOneScalarTail (const VectorizationFactor &CurrentFactor, bool HasTail,
+ bool ForceVectorization, const ElementCount &ExactTC,
+ const VectorizationFactor &ScalarFactor, unsigned int UserIC);
+
/// Compute and return the most profitable vectorization factor and the
/// corresponding best VPlan. Also collect all profitable VFs in
/// ProfitableVFs.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 7a975351cb12e..612d861e64d88 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3082,7 +3082,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
if ((TC - 1) % EffectiveIC == 0) {
unsigned VF = (TC - 1) / EffectiveIC;
if (VF >= 2 && VF <= MaxFixedVF && isPowerOf2_32(VF)) {
- LLVM_DEBUG(dbgs() << "LV: Picking VF=" << VF
+ LLVM_DEBUG(dbgs() << "LV: Picking MaxVF=" << VF
<< " with 1 scalar iteration remaining.\n");
MaxFactors.FixedVF = ElementCount::getFixed(VF);
MaxFactors.ScalableVF = ElementCount::getScalable(0);
@@ -5827,48 +5827,49 @@ InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
return Cost;
}
+bool LoopVectorizationPlanner::isUnprofitableOneScalarTail (const VectorizationFactor &CurrentFactor, bool HasTail,
+ bool ForceVectorization, const ElementCount &ExactTC,
+ const VectorizationFactor &ScalarFactor, unsigned int UserIC) {
+ if (ForceVectorization || !HasTail || !ExactTC.isFixed() ||
+ CurrentFactor.Width.isScalable())
+ return false;
+
+ unsigned TC = ExactTC.getFixedValue();
+ if (TC == 0 || TC > TTI.getMinTripCountTailFoldingThreshold())
+ return false;
+
+ unsigned EstimatedWidth = estimateElementCount(
+ CurrentFactor.Width, Config.getVScaleForTuning());
+ if (TC != (EstimatedWidth * UserIC) + 1)
+ return false;
+
+ InstructionCost VectorCost =
+ getCostForKnownTripCount(CurrentFactor, TC, /*HasTail=*/true);
+ InstructionCost ScalarCostForTC =
+ getCostForKnownTripCount(ScalarFactor, TC, /*HasTail=*/false);
+ // Be conservative for the one-scalar-tail shape. It introduces extra
+ // control flow and a scalar epilogue for a single element, so require
+ // the vectorized form to save at least one scalar iteration.
+ InstructionCost AdjustedVectorCost = VectorCost + ScalarFactor.ScalarCost;
+ if (!VectorCost.isValid() ||
+ AdjustedVectorCost < ScalarCostForTC) {
+ LLVM_DEBUG(dbgs() << "LV: Accepting VF " << CurrentFactor.Width << " for one-scalar-tail low trip count: vector cost " << AdjustedVectorCost << " < " << ScalarCostForTC << ".\n");
+ return false;
+ }
+
+ LLVM_DEBUG(dbgs() << "LV: Rejecting VF " << CurrentFactor.Width
+ << " for one-scalar-tail low trip count: vector cost "
+ << AdjustedVectorCost << " >= scalar cost "
+ << ScalarCostForTC << ".\n");
+ return true;
+}
+
std::pair<VectorizationFactor, VPlan *>
LoopVectorizationPlanner::computeBestVF() {
if (VPlans.empty())
return {VectorizationFactor::Disabled(), nullptr};
// If there is a single VPlan with a single VF, return it directly.
VPlan &FirstPlan = *VPlans[0];
- auto IsUnprofitableOneScalarTail =
- [&](const VectorizationFactor &CurrentFactor, bool HasTail,
- bool ForceVectorization, const ElementCount &ExactTC,
- const VectorizationFactor &ScalarFactor,
- const InstructionCost &ScalarCost, unsigned int UserIC) {
- if (ForceVectorization || !HasTail || !ExactTC.isFixed() ||
- CurrentFactor.Width.isScalable())
- return false;
-
- unsigned TC = ExactTC.getFixedValue();
- if (TC == 0 || TC > TTI.getMinTripCountTailFoldingThreshold())
- return false;
-
- unsigned EstimatedWidth = estimateElementCount(
- CurrentFactor.Width, Config.getVScaleForTuning());
- if (TC != (EstimatedWidth * UserIC) + 1)
- return false;
-
- InstructionCost VectorCost =
- getCostForKnownTripCount(CurrentFactor, TC, HasTail);
- InstructionCost ScalarCostForTC =
- getCostForKnownTripCount(ScalarFactor, TC, /*HasTail=*/false);
- // Be conservative for the one-scalar-tail shape. It introduces extra
- // control flow and a scalar epilogue for a single element, so require
- // the vectorized form to save at least one scalar iteration.
- InstructionCost AdjustedVectorCost = VectorCost + ScalarCost;
- if (!AdjustedVectorCost.isValid() ||
- AdjustedVectorCost < ScalarCostForTC)
- return false;
-
- LLVM_DEBUG(dbgs() << "LV: Rejecting VF " << CurrentFactor.Width
- << " for one-scalar-tail low trip count: vector cost "
- << AdjustedVectorCost << " >= scalar cost "
- << ScalarCostForTC << ".\n");
- return true;
- };
ElementCount UserVF = Config.getHints().getWidth();
unsigned int UserIC = Config.getHints().getInterleave() != 0 ? Config.getHints().getInterleave() : 1;
@@ -5984,9 +5985,9 @@ LoopVectorizationPlanner::computeBestVF() {
cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
- if (IsUnprofitableOneScalarTail(CurrentFactor, P->hasScalarTail(),
- ForceVectorization, ExactTC, ScalarFactor,
- ScalarCost, UserIC))
+ unsigned int UserIC = Hints.getInterleave() != 0 ? Hints.getInterleave() : 1;
+ if (isUnprofitableOneScalarTail(CurrentFactor, P->hasScalarTail(),
+ ForceVectorization, ExactTC, ScalarFactor, UserIC))
continue;
if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
index 70eb5b07c25aa..4516062f9b7d1 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
@@ -10,7 +10,7 @@ define void @tc3_udiv_i8_reject(ptr noalias %a, ptr noalias %b,
; IR-NOT: vector.body:
;
; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_reject'
-; DBG: LV: Picking VF=2 with 1 scalar iteration remaining.
+; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
; DBG: LV: Scalar loop costs: 9.
; DBG: Cost for VF 2: 19
; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 37 >= scalar cost 27.
@@ -36,43 +36,12 @@ exit:
ret void
}
-define void @tc3_udiv_i8_user_vf2(ptr noalias %a, ptr noalias %b,
- ptr noalias %c) #0 {
-; IR-LABEL: define void @tc3_udiv_i8_user_vf2(
-; IR-NOT: vector.body
-
-; DBG-LABEL: LV: Checking a loop in 'tc3_udiv_i8_user_vf2'
-; DBG: LV: Using user VF 2.
-; DBG: LV: Scalar loop costs: 9.
-; DBG: Cost for VF 2: 19
-; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 37 >= scalar cost 27.
-; DBG: LV: Vectorization is possible but not beneficial.
-entry:
- br label %loop
-
-loop:
- %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
- %pa = getelementptr inbounds i8, ptr %a, i64 %iv
- %pb = getelementptr inbounds i8, ptr %b, i64 %iv
- %pc = getelementptr inbounds i8, ptr %c, i64 %iv
- %va = load i8, ptr %pa, align 1
- %vb = load i8, ptr %pb, align 1
- %div = udiv i8 %va, %vb
- store i8 %div, ptr %pc, align 1
- %iv.next = add nuw nsw i64 %iv, 1
- %exitcond = icmp eq i64 %iv.next, 3
- br i1 %exitcond, label %exit, label %loop, !llvm.loop !2
-
-exit:
- ret void
-}
-
define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
; IR-LABEL: define void @tc3_smin_i8_reject(
; IR-NOT: vector.body
; DBG-LABEL: LV: Checking a loop in 'tc3_smin_i8_reject'
-; DBG: LV: Picking VF=2 with 1 scalar iteration remaining.
+; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
; DBG: LV: Scalar loop costs: 10.
; DBG: Cost for VF 2: 15
; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 35 >= scalar cost 30.
@@ -131,7 +100,7 @@ define void @tc5_udiv_i8_reject_ic2(ptr noalias %a, ptr noalias %b, ptr noalias
; DBG-LABEL: LV: Checking a loop in 'tc5_udiv_i8_reject_ic2'
; DBG-NOT: LV: Selecting VF: 2.
-; DBG Rejecting VF 2
+; DBG: Rejecting VF 2
entry:
br label %loop
@@ -146,7 +115,37 @@ loop:
store i8 %div, ptr %pc, align 1
%iv.next = add nuw nsw i64 %iv, 1
%exitcond = icmp eq i64 %iv.next, 5
- br i1 %exitcond, label %exit, label %loop, !llvm.loop !4
+ br i1 %exitcond, label %exit, label %loop, !llvm.loop !2
+
+exit:
+ ret void
+}
+
+define void @tc5_sin_f32_select_smaller_vf(ptr noalias %a,
+ ptr noalias %c) #0 {
+; IR-LABEL: define void @tc5_sin_f32_select_smaller_vf(
+; IR: vector.body
+
+; DBG-LABEL: LV: Checking a loop in 'tc5_sin_f32_select_smaller_vf'
+; DBG: Picking MaxVF=4 with 1 scalar iteration remaining.
+; DBG: LV: Rejecting VF 4 for one-scalar-tail low trip count: vector cost 88 >= scalar cost 80.
+; DBG-NOT: Selecting VF: 4
+; DBG: LV: Selecting VF: 2.
+
+
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %pa = getelementptr inbounds float, ptr %a, i64 %iv
+ %pc = getelementptr inbounds float, ptr %c, i64 %iv
+ %va = load float, ptr %pa, align 4
+ %sin = tail call float @llvm.sin.f32(float %va)
+ store float %sin, ptr %pc, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop
exit:
ret void
@@ -154,11 +153,11 @@ exit:
declare i8 @llvm.smin.i8(i8, i8)
+declare float @llvm.sin.f32(float)
+
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
!0 = distinct !{!0, !1}
!1 = !{!"llvm.loop.vectorize.enable", i1 true}
!2 = distinct !{!2, !3}
-!3 = !{!"llvm.loop.vectorize.width", i32 2}
-!4 = distinct !{!4, !5}
-!5 = !{!"llvm.loop.interleave.count", i32 2}
+!3 = !{!"llvm.loop.interleave.count", i32 2}
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
index 5c08f97a55aa2..b999acc21309c 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -375,6 +375,87 @@ exit:
ret void
}
+; For this example, vectorization should not take place as the
+; scalar cost is better than the vectorized cost.
+define void @tc3_udiv_i8_reject(ptr noalias %a, ptr noalias %b,
+; CHECK-LABEL: define void @tc3_udiv_i8_reject(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[PA:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[PB:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[PC:%.*]] = getelementptr inbounds i8, ptr [[C]], i64 [[IV]]
+; CHECK-NEXT: [[VA:%.*]] = load i8, ptr [[PA]], align 1
+; CHECK-NEXT: [[VB:%.*]] = load i8, ptr [[PB]], align 1
+; CHECK-NEXT: [[DIV:%.*]] = udiv i8 [[VA]], [[VB]]
+; CHECK-NEXT: store i8 [[DIV]], ptr [[PC]], align 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 3
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+ ptr noalias %c) #0 {
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %pa = getelementptr inbounds i8, ptr %a, i64 %iv
+ %pb = getelementptr inbounds i8, ptr %b, i64 %iv
+ %pc = getelementptr inbounds i8, ptr %c, i64 %iv
+ %va = load i8, ptr %pa, align 1
+ %vb = load i8, ptr %pb, align 1
+ %div = udiv i8 %va, %vb
+ store i8 %div, ptr %pc, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc3_smin_i8_reject(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[TMP1:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
+; CHECK-NEXT: [[ARRAYIDX2:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[TMP2:%.*]] = load i8, ptr [[ARRAYIDX2]], align 1
+; CHECK-NEXT: [[MIN:%.*]] = tail call i8 @llvm.smin.i8(i8 [[TMP1]], i8 [[TMP2]])
+; CHECK-NEXT: store i8 [[MIN]], ptr [[ARRAYIDX]], align 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 3
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %arrayidx = getelementptr inbounds i8, ptr %b, i64 %iv
+ %0 = load i8, ptr %arrayidx, align 1
+ %arrayidx2 = getelementptr inbounds i8, ptr %a, i64 %iv
+ %1 = load i8, ptr %arrayidx2, align 1
+ %min = tail call i8 @llvm.smin.i8(i8 %0, i8 %1)
+ store i8 %min, ptr %arrayidx, align 1
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 3
+ br i1 %exitcond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
!0 = distinct !{!0, !1}
>From 0558b9b488c8118e74906ff0b5062bfe1f634c2b Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 16 Jul 2026 15:55:34 +0100
Subject: [PATCH 17/32] format
---
.../Vectorize/LoopVectorizationPlanner.h | 14 +++++----
.../Transforms/Vectorize/LoopVectorize.cpp | 29 +++++++++++--------
2 files changed, 25 insertions(+), 18 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 26cee9e7768ce..7afaf94b23ba0 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -919,12 +919,14 @@ class LoopVectorizationPlanner {
/// for each VF.
VPlan &getPlanFor(ElementCount VF) const;
- /// Examines if it is unprofitable to Vectorize a small loop in a way that leaves a
- /// Vector iteration, followed by a single iteration scalar tail. For some uses cases,
- /// it is better to leave the original Scalar loop in place.
- bool isUnprofitableOneScalarTail (const VectorizationFactor &CurrentFactor, bool HasTail,
- bool ForceVectorization, const ElementCount &ExactTC,
- const VectorizationFactor &ScalarFactor, unsigned int UserIC);
+ /// Examines if it is unprofitable to Vectorize a small loop in a way that
+ /// leaves a Vector iteration, followed by a single iteration scalar tail. For
+ /// some uses cases, it is better to leave the original Scalar loop in place.
+ bool isUnprofitableOneScalarTail(const VectorizationFactor &CurrentFactor,
+ bool HasTail, bool ForceVectorization,
+ const ElementCount &ExactTC,
+ const VectorizationFactor &ScalarFactor,
+ unsigned int UserIC);
/// Compute and return the most profitable vectorization factor and the
/// corresponding best VPlan. Also collect all profitable VFs in
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 612d861e64d88..9698dd417a535 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5827,9 +5827,10 @@ InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
return Cost;
}
-bool LoopVectorizationPlanner::isUnprofitableOneScalarTail (const VectorizationFactor &CurrentFactor, bool HasTail,
- bool ForceVectorization, const ElementCount &ExactTC,
- const VectorizationFactor &ScalarFactor, unsigned int UserIC) {
+bool LoopVectorizationPlanner::isUnprofitableOneScalarTail(
+ const VectorizationFactor &CurrentFactor, bool HasTail,
+ bool ForceVectorization, const ElementCount &ExactTC,
+ const VectorizationFactor &ScalarFactor, unsigned int UserIC) {
if (ForceVectorization || !HasTail || !ExactTC.isFixed() ||
CurrentFactor.Width.isScalable())
return false;
@@ -5838,8 +5839,8 @@ bool LoopVectorizationPlanner::isUnprofitableOneScalarTail (const VectorizationF
if (TC == 0 || TC > TTI.getMinTripCountTailFoldingThreshold())
return false;
- unsigned EstimatedWidth = estimateElementCount(
- CurrentFactor.Width, Config.getVScaleForTuning());
+ unsigned EstimatedWidth =
+ estimateElementCount(CurrentFactor.Width, Config.getVScaleForTuning());
if (TC != (EstimatedWidth * UserIC) + 1)
return false;
@@ -5851,11 +5852,13 @@ bool LoopVectorizationPlanner::isUnprofitableOneScalarTail (const VectorizationF
// control flow and a scalar epilogue for a single element, so require
// the vectorized form to save at least one scalar iteration.
InstructionCost AdjustedVectorCost = VectorCost + ScalarFactor.ScalarCost;
- if (!VectorCost.isValid() ||
- AdjustedVectorCost < ScalarCostForTC) {
- LLVM_DEBUG(dbgs() << "LV: Accepting VF " << CurrentFactor.Width << " for one-scalar-tail low trip count: vector cost " << AdjustedVectorCost << " < " << ScalarCostForTC << ".\n");
- return false;
- }
+ if (!VectorCost.isValid() || AdjustedVectorCost < ScalarCostForTC) {
+ LLVM_DEBUG(dbgs() << "LV: Accepting VF " << CurrentFactor.Width
+ << " for one-scalar-tail low trip count: vector cost "
+ << AdjustedVectorCost << " < " << ScalarCostForTC
+ << ".\n");
+ return false;
+ }
LLVM_DEBUG(dbgs() << "LV: Rejecting VF " << CurrentFactor.Width
<< " for one-scalar-tail low trip count: vector cost "
@@ -5985,9 +5988,11 @@ LoopVectorizationPlanner::computeBestVF() {
cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
- unsigned int UserIC = Hints.getInterleave() != 0 ? Hints.getInterleave() : 1;
+ unsigned int UserIC =
+ Hints.getInterleave() != 0 ? Hints.getInterleave() : 1;
if (isUnprofitableOneScalarTail(CurrentFactor, P->hasScalarTail(),
- ForceVectorization, ExactTC, ScalarFactor, UserIC))
+ ForceVectorization, ExactTC, ScalarFactor,
+ UserIC))
continue;
if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
>From 9d291bbe41f845289705debed54a3c916990fb9d Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 20 Jul 2026 10:04:12 +0100
Subject: [PATCH 18/32] Respond to review comments
---
.../Vectorize/LoopVectorizationPlanner.h | 13 +++---
.../Transforms/Vectorize/LoopVectorize.cpp | 42 ++++++++-----------
.../sve-small-trip-count-vf-plus-one-cost.ll | 8 ++--
.../sve-small-trip-count-vf-plus-one.ll | 24 ++++++-----
4 files changed, 42 insertions(+), 45 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 7afaf94b23ba0..a71e171343ab1 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -920,13 +920,12 @@ class LoopVectorizationPlanner {
VPlan &getPlanFor(ElementCount VF) const;
/// Examines if it is unprofitable to Vectorize a small loop in a way that
- /// leaves a Vector iteration, followed by a single iteration scalar tail. For
- /// some uses cases, it is better to leave the original Scalar loop in place.
- bool isUnprofitableOneScalarTail(const VectorizationFactor &CurrentFactor,
- bool HasTail, bool ForceVectorization,
- const ElementCount &ExactTC,
- const VectorizationFactor &ScalarFactor,
- unsigned int UserIC);
+ /// leaves a Vector iteration, followed by a single iteration scalar tail.
+ /// Returns true if it is profitable to vectorize a small loop with a single
+ /// scalar tail iteration
+ bool isProfitableOneScalarTail(const VectorizationFactor &CurrentFactor,
+ const ElementCount &ExactTC,
+ unsigned int UserIC);
/// Compute and return the most profitable vectorization factor and the
/// corresponding best VPlan. Also collect all profitable VFs in
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 9698dd417a535..498075b439db9 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5827,44 +5827,39 @@ InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
return Cost;
}
-bool LoopVectorizationPlanner::isUnprofitableOneScalarTail(
- const VectorizationFactor &CurrentFactor, bool HasTail,
- bool ForceVectorization, const ElementCount &ExactTC,
- const VectorizationFactor &ScalarFactor, unsigned int UserIC) {
- if (ForceVectorization || !HasTail || !ExactTC.isFixed() ||
- CurrentFactor.Width.isScalable())
- return false;
+bool LoopVectorizationPlanner::isProfitableOneScalarTail(
+ const VectorizationFactor &CurrentFactor, const ElementCount &ExactTC,
+ unsigned int UserIC) {
+ if (!ExactTC.isFixed() || CurrentFactor.Width.isScalable())
+ return true;
unsigned TC = ExactTC.getFixedValue();
if (TC == 0 || TC > TTI.getMinTripCountTailFoldingThreshold())
- return false;
+ return true;
unsigned EstimatedWidth =
estimateElementCount(CurrentFactor.Width, Config.getVScaleForTuning());
if (TC != (EstimatedWidth * UserIC) + 1)
- return false;
+ return true;
+ // VectorCost reflects the cost of the requried vector iteration(s) and the
+ // one remaining scalar iteration cost
InstructionCost VectorCost =
getCostForKnownTripCount(CurrentFactor, TC, /*HasTail=*/true);
- InstructionCost ScalarCostForTC =
- getCostForKnownTripCount(ScalarFactor, TC, /*HasTail=*/false);
- // Be conservative for the one-scalar-tail shape. It introduces extra
- // control flow and a scalar epilogue for a single element, so require
- // the vectorized form to save at least one scalar iteration.
- InstructionCost AdjustedVectorCost = VectorCost + ScalarFactor.ScalarCost;
- if (!VectorCost.isValid() || AdjustedVectorCost < ScalarCostForTC) {
+ InstructionCost ScalarCostForTC = CurrentFactor.ScalarCost * TC;
+ if (!VectorCost.isValid() || VectorCost < ScalarCostForTC) {
LLVM_DEBUG(dbgs() << "LV: Accepting VF " << CurrentFactor.Width
<< " for one-scalar-tail low trip count: vector cost "
- << AdjustedVectorCost << " < " << ScalarCostForTC
+ << VectorCost << " < scalar cost " << ScalarCostForTC
<< ".\n");
- return false;
+ return true;
}
LLVM_DEBUG(dbgs() << "LV: Rejecting VF " << CurrentFactor.Width
<< " for one-scalar-tail low trip count: vector cost "
- << AdjustedVectorCost << " >= scalar cost "
- << ScalarCostForTC << ".\n");
- return true;
+ << VectorCost << " >= scalar cost " << ScalarCostForTC
+ << ".\n");
+ return false;
}
std::pair<VectorizationFactor, VPlan *>
@@ -5990,9 +5985,8 @@ LoopVectorizationPlanner::computeBestVF() {
unsigned int UserIC =
Hints.getInterleave() != 0 ? Hints.getInterleave() : 1;
- if (isUnprofitableOneScalarTail(CurrentFactor, P->hasScalarTail(),
- ForceVectorization, ExactTC, ScalarFactor,
- UserIC))
+ if (!ForceVectorization && P->hasScalarTail() &&
+ !isProfitableOneScalarTail(CurrentFactor, ExactTC, UserIC))
continue;
if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
index 4516062f9b7d1..e877af9dc88e8 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
@@ -13,7 +13,7 @@ define void @tc3_udiv_i8_reject(ptr noalias %a, ptr noalias %b,
; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
; DBG: LV: Scalar loop costs: 9.
; DBG: Cost for VF 2: 19
-; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 37 >= scalar cost 27.
+; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 47 >= scalar cost 27.
; DBG: LV: Selecting VF: 1.
; DBG: LV: Vectorization is possible but not beneficial.
entry:
@@ -44,7 +44,7 @@ define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
; DBG: LV: Scalar loop costs: 10.
; DBG: Cost for VF 2: 15
-; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 35 >= scalar cost 30.
+; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 40 >= scalar cost 30.
; DBG: LV: Selecting VF: 1.
; DBG: LV: Vectorization is possible but not beneficial.
entry:
@@ -100,7 +100,7 @@ define void @tc5_udiv_i8_reject_ic2(ptr noalias %a, ptr noalias %b, ptr noalias
; DBG-LABEL: LV: Checking a loop in 'tc5_udiv_i8_reject_ic2'
; DBG-NOT: LV: Selecting VF: 2.
-; DBG: Rejecting VF 2
+; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 85 >= scalar cost 45.
entry:
br label %loop
@@ -128,7 +128,7 @@ define void @tc5_sin_f32_select_smaller_vf(ptr noalias %a,
; DBG-LABEL: LV: Checking a loop in 'tc5_sin_f32_select_smaller_vf'
; DBG: Picking MaxVF=4 with 1 scalar iteration remaining.
-; DBG: LV: Rejecting VF 4 for one-scalar-tail low trip count: vector cost 88 >= scalar cost 80.
+; DBG: LV: Accepting VF 4 for one-scalar-tail low trip count: vector cost 72 < scalar cost 80.
; DBG-NOT: Selecting VF: 4
; DBG: LV: Selecting VF: 2.
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
index b999acc21309c..dc947ff778280 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -106,9 +106,8 @@ exit:
ret void
}
-; TC=5, VF=4: TC == VF + 1 (5 == 4 + 1).
; VF=16 is a natural fixed-width for i8 types on AArch64, so this checks that the
-; search can step down more than once before accepting VF=4.
+; search can step down more than once before accepting the most acceptable VF.
define void @tc5_vf4_vectorize_i8(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_vectorize_i8(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
@@ -117,10 +116,15 @@ define void @tc5_vf4_vectorize_i8(ptr noalias %a, ptr noalias %b) #0 {
; CHECK: [[LOOP]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[A]], align 1
-; CHECK-NEXT: [[TMP0:%.*]] = add <4 x i8> [[WIDE_LOAD]], splat (i8 1)
-; CHECK-NEXT: store <4 x i8> [[TMP0]], ptr [[B]], align 1
-; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[LOOP]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i8>, ptr [[TMP0]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = add <2 x i8> [[WIDE_LOAD]], splat (i8 1)
+; CHECK-NEXT: store <2 x i8> [[TMP2]], ptr [[TMP1]], align 1
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 2
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4
+; CHECK-NEXT: br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
; CHECK: [[SCALAR_PH]]:
@@ -134,7 +138,7 @@ define void @tc5_vf4_vectorize_i8(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i8 [[ADD]], ptr [[GEP_B]], align 1
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -188,7 +192,7 @@ define void @tc5_forced_ic2_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP6:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -242,7 +246,7 @@ define void @tc5_forced_ic2_vectorize_i64(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i64 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP7:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -351,7 +355,7 @@ define void @tc5_vf4_unsafe_useric_distance4_i32(ptr noalias %a, ptr noalias %b)
; CHECK-NEXT: store i32 [[ADD]], ptr [[B_DST]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 9
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
>From 816d703948b9e3528fd755b4bceed3004abfa4f6 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Tue, 21 Jul 2026 13:23:18 +0100
Subject: [PATCH 19/32] Fix test failures
---
.../sve-small-trip-count-vf-plus-one-cost.ll | 14 +-
.../sve-small-trip-count-vf-plus-one.ll | 142 ++++++++++--------
2 files changed, 84 insertions(+), 72 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
index e877af9dc88e8..5c7da2b49df27 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
@@ -13,7 +13,7 @@ define void @tc3_udiv_i8_reject(ptr noalias %a, ptr noalias %b,
; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
; DBG: LV: Scalar loop costs: 9.
; DBG: Cost for VF 2: 19
-; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 47 >= scalar cost 27.
+; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 28 >= scalar cost 27.
; DBG: LV: Selecting VF: 1.
; DBG: LV: Vectorization is possible but not beneficial.
entry:
@@ -36,17 +36,19 @@ exit:
ret void
}
+; FIXME: This is currently accepted as cost for vector is smaller than scalar.
+; Performance is poor due to poor CodeGen of using type promotion instead of
+; type widening.
define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
; IR-LABEL: define void @tc3_smin_i8_reject(
-; IR-NOT: vector.body
+; IR: vector.body
; DBG-LABEL: LV: Checking a loop in 'tc3_smin_i8_reject'
; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
; DBG: LV: Scalar loop costs: 10.
; DBG: Cost for VF 2: 15
-; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 40 >= scalar cost 30.
-; DBG: LV: Selecting VF: 1.
-; DBG: LV: Vectorization is possible but not beneficial.
+; DBG: LV: Accepting VF 2 for one-scalar-tail low trip count: vector cost 25 < scalar cost 30.
+; DBG: LV: Selecting VF: 2.
entry:
br label %loop
@@ -100,7 +102,7 @@ define void @tc5_udiv_i8_reject_ic2(ptr noalias %a, ptr noalias %b, ptr noalias
; DBG-LABEL: LV: Checking a loop in 'tc5_udiv_i8_reject_ic2'
; DBG-NOT: LV: Selecting VF: 2.
-; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 85 >= scalar cost 45.
+; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 47 >= scalar cost 45.
entry:
br label %loop
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
index dc947ff778280..f26f1fdc7ccce 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -12,9 +12,9 @@ target triple = "aarch64-unknown-linux-gnu"
define void @tc5_vf4_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_vectorize_i32(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[SCALAR_PH1:.*:]]
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
@@ -22,11 +22,11 @@ define void @tc5_vf4_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[B]], align 4
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
-; CHECK: [[SCALAR_PH]]:
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
@@ -34,7 +34,7 @@ define void @tc5_vf4_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -62,9 +62,9 @@ exit:
define void @tc5_vf4_vectorize_i16(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_vectorize_i16(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[A]], align 2
@@ -72,11 +72,11 @@ define void @tc5_vf4_vectorize_i16(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <4 x i16> [[TMP0]], ptr [[B]], align 2
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
-; CHECK: [[SCALAR_PH]]:
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i16, ptr [[GEP_A]], align 2
@@ -84,7 +84,7 @@ define void @tc5_vf4_vectorize_i16(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i16 [[ADD]], ptr [[GEP_B]], align 2
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -111,26 +111,21 @@ exit:
define void @tc5_vf4_vectorize_i8(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_vectorize_i8(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[LOOP]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[INDEX]]
-; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[INDEX]]
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i8>, ptr [[TMP0]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = add <2 x i8> [[WIDE_LOAD]], splat (i8 1)
-; CHECK-NEXT: store <2 x i8> [[TMP2]], ptr [[TMP1]], align 1
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 2
-; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i64 [[INDEX_NEXT]], 4
-; CHECK-NEXT: br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[A]], align 1
+; CHECK-NEXT: [[TMP0:%.*]] = add <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; CHECK-NEXT: store <4 x i8> [[TMP0]], ptr [[B]], align 1
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
-; CHECK: [[SCALAR_PH]]:
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i8, ptr [[GEP_A]], align 1
@@ -138,7 +133,7 @@ define void @tc5_vf4_vectorize_i8(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i8 [[ADD]], ptr [[GEP_B]], align 1
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -165,9 +160,9 @@ exit:
define void @tc5_forced_ic2_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_forced_ic2_vectorize_i32(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH1:.*:]]
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 2
@@ -180,11 +175,11 @@ define void @tc5_forced_ic2_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <2 x i32> [[TMP2]], ptr [[TMP3]], align 4
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
-; CHECK: [[SCALAR_PH]]:
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
@@ -192,7 +187,7 @@ define void @tc5_forced_ic2_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -219,9 +214,9 @@ exit:
define void @tc5_forced_ic2_vectorize_i64(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_forced_ic2_vectorize_i64(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*:]]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 2
@@ -234,11 +229,11 @@ define void @tc5_forced_ic2_vectorize_i64(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <2 x i64> [[TMP2]], ptr [[TMP3]], align 4
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i64, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i64, ptr [[GEP_A]], align 4
@@ -246,7 +241,7 @@ define void @tc5_forced_ic2_vectorize_i64(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i64 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP6:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -327,9 +322,9 @@ exit:
define void @tc5_vf4_unsafe_useric_distance4_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_unsafe_useric_distance4_i32(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ENTRY:.*:]]
-; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
-; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 4
@@ -341,11 +336,11 @@ define void @tc5_vf4_unsafe_useric_distance4_i32(ptr noalias %a, ptr noalias %b)
; CHECK-NEXT: store <4 x i32> [[TMP3]], ptr [[TMP0]], align 4
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
-; CHECK: [[SCALAR_PH]]:
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 8, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 8, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
; CHECK-NEXT: [[B_DST:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[B_SRC:%.*]] = getelementptr inbounds i32, ptr [[B_DST]], i64 -4
; CHECK-NEXT: [[DEP:%.*]] = load i32, ptr [[B_SRC]], align 4
@@ -355,7 +350,7 @@ define void @tc5_vf4_unsafe_useric_distance4_i32(ptr noalias %a, ptr noalias %b)
; CHECK-NEXT: store i32 [[ADD]], ptr [[B_DST]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 9
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP7:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -422,13 +417,28 @@ exit:
ret void
}
+; FIXME: This is currently accepted as cost for vector is smaller than scalar.
+; Performance is poor due to poor CodeGen of using type promotion instead of
+; type widening.
define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc3_smin_i8_reject(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i8>, ptr [[B]], align 1
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <2 x i8>, ptr [[A]], align 1
+; CHECK-NEXT: [[TMP0:%.*]] = call <2 x i8> @llvm.smin.v2i8(<2 x i8> [[WIDE_LOAD]], <2 x i8> [[WIDE_LOAD1]])
+; CHECK-NEXT: store <2 x i8> [[TMP0]], ptr [[B]], align 1
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 2, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
; CHECK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[TMP1:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
; CHECK-NEXT: [[ARRAYIDX2:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
@@ -437,7 +447,7 @@ define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i8 [[MIN]], ptr [[ARRAYIDX]], align 1
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 3
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP8:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
>From 34870db70c2ef0bfc3cf80d728a39df57923f793 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Fri, 7 Aug 2026 09:28:15 +0100
Subject: [PATCH 20/32] Update test case name and fix failing tests
---
.../AArch64/sve-small-trip-count-vf-plus-one-cost.ll | 11 ++++-------
1 file changed, 4 insertions(+), 7 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
index 5c7da2b49df27..d34b1d2585fff 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
@@ -36,14 +36,11 @@ exit:
ret void
}
-; FIXME: This is currently accepted as cost for vector is smaller than scalar.
-; Performance is poor due to poor CodeGen of using type promotion instead of
-; type widening.
-define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
-; IR-LABEL: define void @tc3_smin_i8_reject(
+define void @tc3_smin_i8_accept(ptr noalias %a, ptr noalias %b) #0 {
+; IR-LABEL: define void @tc3_smin_i8_accept(
; IR: vector.body
-; DBG-LABEL: LV: Checking a loop in 'tc3_smin_i8_reject'
+; DBG-LABEL: LV: Checking a loop in 'tc3_smin_i8_accept'
; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
; DBG: LV: Scalar loop costs: 10.
; DBG: Cost for VF 2: 15
@@ -160,6 +157,6 @@ declare float @llvm.sin.f32(float)
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
!0 = distinct !{!0, !1}
-!1 = !{!"llvm.loop.vectorize.enable", i1 true}
+!1 = !{!"llvm.loop.vectorize.enable"}
!2 = distinct !{!2, !3}
!3 = !{!"llvm.loop.interleave.count", i32 2}
>From 73eb2dc08d7bc987a433d9241eda718c63e4445e Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 13 Aug 2026 15:45:20 +0100
Subject: [PATCH 21/32] Add minsize/optsize tests
---
.../sve-small-trip-count-vf-plus-one.ll | 97 ++++++++++++++++++-
1 file changed, 96 insertions(+), 1 deletion(-)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
index f26f1fdc7ccce..8771944dd9dc4 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -8,7 +8,6 @@
target triple = "aarch64-unknown-linux-gnu"
; TC=5, VF=4: TC == MaxFixedVF + 1 (5 == 4 + 1).
-; The new code path should trigger: 1 vectorized iteration + 1 scalar iteration.
define void @tc5_vf4_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_vectorize_i32(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0:[0-9]+]] {
@@ -470,7 +469,103 @@ exit:
ret void
}
+define void @tc5_v4i32_reject_minsize(ptr noalias %a, ptr noalias %b) #1 {
+; CHECK-LABEL: define void @tc5_v4i32_reject_minsize(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR1:[0-9]+]] {
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[B]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[VAL]], 1
+; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP9:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ %val = load i32, ptr %gep.a, align 4
+ %add = add nsw i32 %val, 1
+ store i32 %add, ptr %gep.b, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define void @tc5_v4i32_reject_optsize(ptr noalias %a, ptr noalias %b) #2 {
+; CHECK-LABEL: define void @tc5_v4i32_reject_optsize(
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR2:[0-9]+]] {
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[B]], align 4
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[VAL]], 1
+; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ %val = load i32, ptr %gep.a, align 4
+ %add = add nsw i32 %val, 1
+ store i32 %add, ptr %gep.b, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond = icmp eq i64 %iv.next, 5
+ br i1 %exitcond, label %exit, label %loop
+
+exit:
+ ret void
+}
+
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
+attributes #1 = { vscale_range(1,16) "target-features"="+sve" minsize }
+attributes #2 = { vscale_range(1,16) "target-features"="+sve" optsize }
!0 = distinct !{!0, !1}
!1 = !{!"llvm.loop.interleave.count", i32 2}
>From ea00b2b07df6743fd984a70f91631dc97c565c94 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Thu, 13 Aug 2026 17:40:43 +0100
Subject: [PATCH 22/32] Update cost test after #212801
---
.../AArch64/sve-small-trip-count-vf-plus-one-cost.ll | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
index d34b1d2585fff..c374f163f807a 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
@@ -43,8 +43,8 @@ define void @tc3_smin_i8_accept(ptr noalias %a, ptr noalias %b) #0 {
; DBG-LABEL: LV: Checking a loop in 'tc3_smin_i8_accept'
; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
; DBG: LV: Scalar loop costs: 10.
-; DBG: Cost for VF 2: 15
-; DBG: LV: Accepting VF 2 for one-scalar-tail low trip count: vector cost 25 < scalar cost 30.
+; DBG: Cost for VF 2: 19
+; DBG: LV: Accepting VF 2 for one-scalar-tail low trip count: vector cost 29 < scalar cost 30.
; DBG: LV: Selecting VF: 2.
entry:
br label %loop
>From 737cf4c812b4942a22d75160343a6f4644e4e13d Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 17 Aug 2026 10:53:38 +0100
Subject: [PATCH 23/32] Ensure transformation is not applied for
minsize/optsize
---
.../Transforms/Vectorize/LoopVectorize.cpp | 6 +++-
.../sve-small-trip-count-vf-plus-one.ll | 34 ++++---------------
2 files changed, 11 insertions(+), 29 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 498075b439db9..22a98fdd85aa1 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3074,9 +3074,13 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// This produces 1 vector iteration, and 1 scalar iteration with
// no remainder. Later passes will eliminate the loop and leave
// straight-line code as the both iteration counts are statically known.
+ //
+ // If a function is marked as minsize/optsize or OptForSize is set, do not
+ // allow this form of transformation as this will increase CodeSize.
ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), TheLoop);
if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
- ExactTC.getFixedValue() > 1) {
+ ExactTC.getFixedValue() > 1 && !TheFunction->hasOptSize() &&
+ !Config.OptForSize) {
unsigned TC = ExactTC.getFixedValue();
unsigned MaxFixedVF = MaxFactors.FixedVF.getFixedValue();
if ((TC - 1) % EffectiveIC == 0) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
index 8771944dd9dc4..4d959504fb7fe 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -472,21 +472,10 @@ exit:
define void @tc5_v4i32_reject_minsize(ptr noalias %a, ptr noalias %b) #1 {
; CHECK-LABEL: define void @tc5_v4i32_reject_minsize(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR1:[0-9]+]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
-; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
-; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
-; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[B]], align 4
-; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
-; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*]]:
; CHECK-NEXT: br label %[[LOOP1:.*]]
; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
@@ -494,7 +483,7 @@ define void @tc5_v4i32_reject_minsize(ptr noalias %a, ptr noalias %b) #1 {
; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP9:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -519,21 +508,10 @@ exit:
define void @tc5_v4i32_reject_optsize(ptr noalias %a, ptr noalias %b) #2 {
; CHECK-LABEL: define void @tc5_v4i32_reject_optsize(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR2:[0-9]+]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
-; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
-; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
-; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[B]], align 4
-; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
-; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*]]:
; CHECK-NEXT: br label %[[LOOP1:.*]]
; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
@@ -541,7 +519,7 @@ define void @tc5_v4i32_reject_optsize(ptr noalias %a, ptr noalias %b) #2 {
; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
>From 3a05a5c915b5bb1556506c4db8cbb2efdf18b11f Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 19 Aug 2026 10:10:29 +0100
Subject: [PATCH 24/32] Ensure TC == (VF * IC) + 1 ensures 1 vector iteration
There are cases where the Loop Vectorizer might pick a VF that has
multiple vector iterations, when 1 iteration is possible. This ensures
this is followed while still considering any Interleave Counts
considered by the user.
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 15 ++++++++++++---
.../sve-small-trip-count-vf-plus-one-cost.ll | 10 ++++++----
2 files changed, 18 insertions(+), 7 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 22a98fdd85aa1..ec3547afc85cd 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5848,6 +5848,9 @@ bool LoopVectorizationPlanner::isProfitableOneScalarTail(
// VectorCost reflects the cost of the requried vector iteration(s) and the
// one remaining scalar iteration cost
+ unsigned VF = CurrentFactor.Width.getFixedValue();
+ if (TC != (VF * UserIC) + 1 && TC != (VF * UserIC))
+ return false;
InstructionCost VectorCost =
getCostForKnownTripCount(CurrentFactor, TC, /*HasTail=*/true);
InstructionCost ScalarCostForTC = CurrentFactor.ScalarCost * TC;
@@ -5988,10 +5991,16 @@ LoopVectorizationPlanner::computeBestVF() {
VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
unsigned int UserIC =
- Hints.getInterleave() != 0 ? Hints.getInterleave() : 1;
+ Config.getHints().getInterleave() != 0 ? Config.getHints().getInterleave() : 1;
if (!ForceVectorization && P->hasScalarTail() &&
- !isProfitableOneScalarTail(CurrentFactor, ExactTC, UserIC))
- continue;
+ isProfitableOneScalarTail(CurrentFactor, ExactTC, UserIC)) {
+ // If we have identified a case where a small trip count loop can
+ // be Vectorized as one Vector Iteration and, if needed, one scalar
+ // iteration, use this VF and break.
+ BestFactor = CurrentFactor;
+ PlanForBestVF = P.get();
+ break;
+ }
if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
BestFactor = CurrentFactor;
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
index c374f163f807a..a201d39b1ad3f 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
@@ -120,16 +120,18 @@ exit:
ret void
}
-define void @tc5_sin_f32_select_smaller_vf(ptr noalias %a,
+; Ensure that the LoopVectorizer will select a VF that will
+; ensure TC == (VF * IC) + 1.
+define void @tc5_sin_f32_dont_select_smaller_vf(ptr noalias %a,
ptr noalias %c) #0 {
; IR-LABEL: define void @tc5_sin_f32_select_smaller_vf(
; IR: vector.body
; DBG-LABEL: LV: Checking a loop in 'tc5_sin_f32_select_smaller_vf'
; DBG: Picking MaxVF=4 with 1 scalar iteration remaining.
-; DBG: LV: Accepting VF 4 for one-scalar-tail low trip count: vector cost 72 < scalar cost 80.
-; DBG-NOT: Selecting VF: 4
-; DBG: LV: Selecting VF: 2.
+; DBG: Accepting VF 4 for one-scalar-tail low trip count: vector cost 72 < scalar cost 80.
+; DBG-NOT: Selecting VF: 2.
+; DBG: Selecting VF: 4
entry:
>From 0e351fd439e998c03acd6404e0c512a86ea4c7af Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 19 Aug 2026 13:55:04 +0100
Subject: [PATCH 25/32] Fix failing test
---
.../AArch64/sve-small-trip-count-vf-plus-one-cost.ll | 4 ++--
1 file changed, 2 insertions(+), 2 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
index a201d39b1ad3f..9fa833a11430e 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
@@ -124,10 +124,10 @@ exit:
; ensure TC == (VF * IC) + 1.
define void @tc5_sin_f32_dont_select_smaller_vf(ptr noalias %a,
ptr noalias %c) #0 {
-; IR-LABEL: define void @tc5_sin_f32_select_smaller_vf(
+; IR-LABEL: define void @tc5_sin_f32_dont_select_smaller_vf(
; IR: vector.body
-; DBG-LABEL: LV: Checking a loop in 'tc5_sin_f32_select_smaller_vf'
+; DBG-LABEL: LV: Checking a loop in 'tc5_sin_f32_dont_select_smaller_vf'
; DBG: Picking MaxVF=4 with 1 scalar iteration remaining.
; DBG: Accepting VF 4 for one-scalar-tail low trip count: vector cost 72 < scalar cost 80.
; DBG-NOT: Selecting VF: 2.
>From e5cab2ea22f1700cb2d64d7d5d6c0ae2d0c01cbd Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Fri, 21 Aug 2026 15:27:51 +0100
Subject: [PATCH 26/32] Restructure PR to simplify changes
This allows LV to work on the costing as it would have already done so.
It simplifies it to just selecting a MaxVF for these type of loops and
then when the TC is below or equal to the tail folding threshold, the
VF is restricted to the VF where VF == TC or VF == (TC - 1)
---
.../Vectorize/LoopVectorizationPlanner.cpp | 42 +++---
.../Vectorize/LoopVectorizationPlanner.h | 13 --
.../Transforms/Vectorize/LoopVectorize.cpp | 130 ++++-------------
.../sve-small-trip-count-vf-plus-one-cost.ll | 4 -
.../sve-small-trip-count-vf-plus-one.ll | 133 +++++++++---------
5 files changed, 110 insertions(+), 212 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index 33802c501033b..cbe2f63f96005 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -768,8 +768,25 @@ bool LoopVectorizationPlanner::isMoreProfitable(const VectorizationFactor &A,
if (!MaxTripCount)
return LowerCostWithoutTC;
- auto RTCostA = getCostForKnownTripCount(A, MaxTripCount, HasTail);
- auto RTCostB = getCostForKnownTripCount(B, MaxTripCount, HasTail);
+ auto GetCostForTC = [MaxTripCount, HasTail](unsigned VF,
+ InstructionCost VectorCost,
+ InstructionCost ScalarCost) {
+ // If the trip count is a known (possibly small) constant, the trip count
+ // will be rounded up to an integer number of iterations under
+ // FoldTailByMasking. The total cost in that case will be
+ // VecCost*ceil(TripCount/VF). When not folding the tail, the total
+ // cost will be VecCost*floor(TC/VF) + ScalarCost*(TC%VF). There will be
+ // some extra overheads, but for the purpose of comparing the costs of
+ // different VFs we can use this to compare the total loop-body cost
+ // expected after vectorization.
+ if (HasTail)
+ return VectorCost * (MaxTripCount / VF) +
+ ScalarCost * (MaxTripCount % VF);
+ return VectorCost * divideCeil(MaxTripCount, VF);
+ };
+
+ auto RTCostA = GetCostForTC(EstimatedWidthA, CostA, A.ScalarCost);
+ auto RTCostB = GetCostForTC(EstimatedWidthB, CostB, B.ScalarCost);
bool LowerCostWithTC = CmpFn(RTCostA, RTCostB);
LLVM_DEBUG(if (LowerCostWithTC != LowerCostWithoutTC) {
dbgs() << "LV: VF " << (LowerCostWithTC ? A.Width : B.Width)
@@ -791,27 +808,6 @@ bool LoopVectorizationPlanner::isMoreProfitable(const VectorizationFactor &A,
IsEpilogue);
}
-InstructionCost LoopVectorizationPlanner::getCostForKnownTripCount(
- const VectorizationFactor &VF, unsigned TripCount, bool HasTail) const {
- unsigned EstimatedWidth = VF.Width.getKnownMinValue();
- if (std::optional<unsigned> VScale = Config.getVScaleForTuning())
- if (VF.Width.isScalable())
- EstimatedWidth *= *VScale;
-
- // If the trip count is a known (possibly small) constant, the trip count
- // will be rounded up to an integer number of iterations under
- // FoldTailByMasking. The total cost in that case will be
- // VecCost*ceil(TripCount/VF). When not folding the tail, the total
- // cost will be VecCost*floor(TC/VF) + ScalarCost*(TC%VF). There will be
- // some extra overheads, but for the purpose of comparing the costs of
- // different VFs we can use this to compare the total loop-body cost
- // expected after vectorization.
- if (HasTail)
- return VF.Cost * (TripCount / EstimatedWidth) +
- VF.ScalarCost * (TripCount % EstimatedWidth);
- return VF.Cost * divideCeil(TripCount, EstimatedWidth);
-}
-
// TODO: we could return a pair of values that specify the max VF and
// min VF, to be used in `buildVPlans(MinVF, MaxVF)` instead of
// `buildVPlans(VF, VF)`. We cannot do it because VPLAN at the moment
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index a71e171343ab1..cca1d4dbc5720 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -919,14 +919,6 @@ class LoopVectorizationPlanner {
/// for each VF.
VPlan &getPlanFor(ElementCount VF) const;
- /// Examines if it is unprofitable to Vectorize a small loop in a way that
- /// leaves a Vector iteration, followed by a single iteration scalar tail.
- /// Returns true if it is profitable to vectorize a small loop with a single
- /// scalar tail iteration
- bool isProfitableOneScalarTail(const VectorizationFactor &CurrentFactor,
- const ElementCount &ExactTC,
- unsigned int UserIC);
-
/// Compute and return the most profitable vectorization factor and the
/// corresponding best VPlan. Also collect all profitable VFs in
/// ProfitableVFs.
@@ -1054,11 +1046,6 @@ class LoopVectorizationPlanner {
const unsigned MaxTripCount, bool HasTail,
bool IsEpilogue = false) const;
- /// Returns the estimated loop-body cost for \p VF and a known trip count.
- InstructionCost getCostForKnownTripCount(const VectorizationFactor &VF,
- unsigned TripCount,
- bool HasTail) const;
-
/// Determines if we have the infrastructure to vectorize the loop and its
/// epilogue, assuming the main loop is vectorized by \p MainPlan.
bool isCandidateForEpilogueVectorization(VPlan &MainPlan) const;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ec3547afc85cd..6b183e3cfc1b7 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3068,32 +3068,27 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
MaxFactors.ScalableVF = ElementCount::getScalable(0);
return MaxFactors;
}
- // Allow cases where the ExactTC == VF + 1. VF can be any power of
- // 2 between 2 and MaxVF.
- //
- // This produces 1 vector iteration, and 1 scalar iteration with
- // no remainder. Later passes will eliminate the loop and leave
- // straight-line code as the both iteration counts are statically known.
- //
- // If a function is marked as minsize/optsize or OptForSize is set, do not
- // allow this form of transformation as this will increase CodeSize.
- ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), TheLoop);
- if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
- ExactTC.getFixedValue() > 1 && !TheFunction->hasOptSize() &&
- !Config.OptForSize) {
- unsigned TC = ExactTC.getFixedValue();
- unsigned MaxFixedVF = MaxFactors.FixedVF.getFixedValue();
- if ((TC - 1) % EffectiveIC == 0) {
- unsigned VF = (TC - 1) / EffectiveIC;
- if (VF >= 2 && VF <= MaxFixedVF && isPowerOf2_32(VF)) {
- LLVM_DEBUG(dbgs() << "LV: Picking MaxVF=" << VF
- << " with 1 scalar iteration remaining.\n");
- MaxFactors.FixedVF = ElementCount::getFixed(VF);
- MaxFactors.ScalableVF = ElementCount::getScalable(0);
- return MaxFactors;
- }
- }
- }
+ }
+
+ // Allow cases where the ExactTC == (VF * IC) + 1.
+ //
+ // This produces 1 vector iteration, and 1 scalar iteration with
+ // no remainder. Later passes will eliminate the loop and leave
+ // straight-line code as the both iteration counts are statically known.
+ //
+ // If a function is marked as minsize/optsize or OptForSize is set, do not
+ // allow this form of transformation as this will increase CodeSize.
+ ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), TheLoop);
+ unsigned TC = ExactTC.getFixedValue();
+ unsigned MyMaxVF = 1ULL << Log2_32(TC);
+ unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
+ if (TC - MyMaxVF == 1 && !TheFunction->hasOptSize() && !Config.OptForSize) {
+ unsigned VF = MyMaxVF / EffectiveIC;
+ LLVM_DEBUG(dbgs() << "LV: Picking MaxVF=" << VF
+ << " with 1 scalar iteration remaining.\n");
+ MaxFactors.FixedVF = ElementCount::getFixed(VF);
+ MaxFactors.ScalableVF = ElementCount::getScalable(0);
+ return MaxFactors;
}
reportVectorizationFailure(
@@ -5831,44 +5826,6 @@ InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
return Cost;
}
-bool LoopVectorizationPlanner::isProfitableOneScalarTail(
- const VectorizationFactor &CurrentFactor, const ElementCount &ExactTC,
- unsigned int UserIC) {
- if (!ExactTC.isFixed() || CurrentFactor.Width.isScalable())
- return true;
-
- unsigned TC = ExactTC.getFixedValue();
- if (TC == 0 || TC > TTI.getMinTripCountTailFoldingThreshold())
- return true;
-
- unsigned EstimatedWidth =
- estimateElementCount(CurrentFactor.Width, Config.getVScaleForTuning());
- if (TC != (EstimatedWidth * UserIC) + 1)
- return true;
-
- // VectorCost reflects the cost of the requried vector iteration(s) and the
- // one remaining scalar iteration cost
- unsigned VF = CurrentFactor.Width.getFixedValue();
- if (TC != (VF * UserIC) + 1 && TC != (VF * UserIC))
- return false;
- InstructionCost VectorCost =
- getCostForKnownTripCount(CurrentFactor, TC, /*HasTail=*/true);
- InstructionCost ScalarCostForTC = CurrentFactor.ScalarCost * TC;
- if (!VectorCost.isValid() || VectorCost < ScalarCostForTC) {
- LLVM_DEBUG(dbgs() << "LV: Accepting VF " << CurrentFactor.Width
- << " for one-scalar-tail low trip count: vector cost "
- << VectorCost << " < scalar cost " << ScalarCostForTC
- << ".\n");
- return true;
- }
-
- LLVM_DEBUG(dbgs() << "LV: Rejecting VF " << CurrentFactor.Width
- << " for one-scalar-tail low trip count: vector cost "
- << VectorCost << " >= scalar cost " << ScalarCostForTC
- << ".\n");
- return false;
-}
-
std::pair<VectorizationFactor, VPlan *>
LoopVectorizationPlanner::computeBestVF() {
if (VPlans.empty())
@@ -5877,40 +5834,12 @@ LoopVectorizationPlanner::computeBestVF() {
VPlan &FirstPlan = *VPlans[0];
ElementCount UserVF = Config.getHints().getWidth();
- unsigned int UserIC = Config.getHints().getInterleave() != 0 ? Config.getHints().getInterleave() : 1;
if (VPlans.size() == 1) {
// For outer loops, the plan has a single vector VF determined by the
// heuristic.
assert((FirstPlan.hasScalarVFOnly() || hasPlanWithVF(UserVF) ||
FirstPlan.isOuterLoop()) &&
"must have a single scalar VF, UserVF or an outer loop");
- bool ForceVectorization =
- Config.getHints().getForce() == LoopVectorizeHints::FK_Enabled;
- if (!FirstPlan.hasScalarVFOnly() && !FirstPlan.isOuterLoop() &&
- hasPlanWithVF(UserVF) && UserVF.isVector() && !ForceVectorization) {
- ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), OrigLoop);
- if (FirstPlan.hasScalarTail() && ExactTC.isFixed() && UserVF.isFixed()) {
- unsigned TC = ExactTC.getFixedValue();
- unsigned EstimatedWidth =
- estimateElementCount(UserVF, Config.getVScaleForTuning());
- if (TC != 0 && TC <= TTI.getMinTripCountTailFoldingThreshold() &&
- TC == EstimatedWidth + 1) {
- ElementCount ScalarVF = ElementCount::getFixed(1);
- InstructionCost ScalarCost = CM.expectedCost(ScalarVF);
- LLVM_DEBUG(dbgs()
- << "LV: Scalar loop costs: " << ScalarCost << ".\n");
-
- InstructionCost Cost = cost(FirstPlan, UserVF, /*RU=*/nullptr);
- VectorizationFactor UserFactor(UserVF, Cost, ScalarCost);
- VectorizationFactor ScalarFactor(ScalarVF, ScalarCost, ScalarCost);
- if (IsUnprofitableOneScalarTail(UserFactor, FirstPlan.hasScalarTail(),
- ForceVectorization, ExactTC,
- ScalarFactor, ScalarCost, UserIC)) {
- return {ScalarFactor, &FirstPlan};
- }
- }
- }
- }
return {VectorizationFactor(FirstPlan.getSingleVF(), 0, 0), &FirstPlan};
}
@@ -5966,6 +5895,11 @@ LoopVectorizationPlanner::computeBestVF() {
if (ConsiderRegPressure)
RUs = calculateRegisterUsageForPlan(*P, VFs, TTI);
+ if (!ForceVectorization && P->hasScalarTail() &&
+ ExactTC.isFixed() && ExactTC.getFixedValue() > 0 && ExactTC.getFixedValue() <= TTI.getMinTripCountTailFoldingThreshold()) {
+ VFs = VFs.take_back(1);
+ }
+
for (unsigned I = 0; I < VFs.size(); I++) {
ElementCount VF = VFs[I];
if (VF.isScalar())
@@ -5990,18 +5924,6 @@ LoopVectorizationPlanner::computeBestVF() {
cost(*P, VF, ConsiderRegPressure ? &RUs[I] : nullptr);
VectorizationFactor CurrentFactor(VF, Cost, ScalarCost);
- unsigned int UserIC =
- Config.getHints().getInterleave() != 0 ? Config.getHints().getInterleave() : 1;
- if (!ForceVectorization && P->hasScalarTail() &&
- isProfitableOneScalarTail(CurrentFactor, ExactTC, UserIC)) {
- // If we have identified a case where a small trip count loop can
- // be Vectorized as one Vector Iteration and, if needed, one scalar
- // iteration, use this VF and break.
- BestFactor = CurrentFactor;
- PlanForBestVF = P.get();
- break;
- }
-
if (isMoreProfitable(CurrentFactor, BestFactor, P->hasScalarTail())) {
BestFactor = CurrentFactor;
PlanForBestVF = P.get();
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
index 9fa833a11430e..634e6e86d9370 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one-cost.ll
@@ -13,7 +13,6 @@ define void @tc3_udiv_i8_reject(ptr noalias %a, ptr noalias %b,
; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
; DBG: LV: Scalar loop costs: 9.
; DBG: Cost for VF 2: 19
-; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 28 >= scalar cost 27.
; DBG: LV: Selecting VF: 1.
; DBG: LV: Vectorization is possible but not beneficial.
entry:
@@ -44,7 +43,6 @@ define void @tc3_smin_i8_accept(ptr noalias %a, ptr noalias %b) #0 {
; DBG: LV: Picking MaxVF=2 with 1 scalar iteration remaining.
; DBG: LV: Scalar loop costs: 10.
; DBG: Cost for VF 2: 19
-; DBG: LV: Accepting VF 2 for one-scalar-tail low trip count: vector cost 29 < scalar cost 30.
; DBG: LV: Selecting VF: 2.
entry:
br label %loop
@@ -99,7 +97,6 @@ define void @tc5_udiv_i8_reject_ic2(ptr noalias %a, ptr noalias %b, ptr noalias
; DBG-LABEL: LV: Checking a loop in 'tc5_udiv_i8_reject_ic2'
; DBG-NOT: LV: Selecting VF: 2.
-; DBG: LV: Rejecting VF 2 for one-scalar-tail low trip count: vector cost 47 >= scalar cost 45.
entry:
br label %loop
@@ -129,7 +126,6 @@ define void @tc5_sin_f32_dont_select_smaller_vf(ptr noalias %a,
; DBG-LABEL: LV: Checking a loop in 'tc5_sin_f32_dont_select_smaller_vf'
; DBG: Picking MaxVF=4 with 1 scalar iteration remaining.
-; DBG: Accepting VF 4 for one-scalar-tail low trip count: vector cost 72 < scalar cost 80.
; DBG-NOT: Selecting VF: 2.
; DBG: Selecting VF: 4
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
index 4d959504fb7fe..97ffae9fa36d1 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-small-trip-count-vf-plus-one.ll
@@ -11,9 +11,9 @@ target triple = "aarch64-unknown-linux-gnu"
define void @tc5_vf4_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_vectorize_i32(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*:]]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A]], align 4
@@ -21,11 +21,11 @@ define void @tc5_vf4_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[B]], align 4
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
@@ -33,7 +33,7 @@ define void @tc5_vf4_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -61,9 +61,9 @@ exit:
define void @tc5_vf4_vectorize_i16(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_vectorize_i16(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[A]], align 2
@@ -71,11 +71,11 @@ define void @tc5_vf4_vectorize_i16(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <4 x i16> [[TMP0]], ptr [[B]], align 2
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i16, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i16, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i16, ptr [[GEP_A]], align 2
@@ -83,7 +83,7 @@ define void @tc5_vf4_vectorize_i16(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i16 [[ADD]], ptr [[GEP_B]], align 2
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -110,9 +110,9 @@ exit:
define void @tc5_vf4_vectorize_i8(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_vectorize_i8(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[A]], align 1
@@ -120,11 +120,11 @@ define void @tc5_vf4_vectorize_i8(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <4 x i8> [[TMP0]], ptr [[B]], align 1
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i8, ptr [[GEP_A]], align 1
@@ -132,7 +132,7 @@ define void @tc5_vf4_vectorize_i8(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i8 [[ADD]], ptr [[GEP_B]], align 1
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -159,9 +159,9 @@ exit:
define void @tc5_forced_ic2_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_forced_ic2_vectorize_i32(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*:]]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 2
@@ -174,11 +174,11 @@ define void @tc5_forced_ic2_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <2 x i32> [[TMP2]], ptr [[TMP3]], align 4
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i32, ptr [[GEP_A]], align 4
@@ -186,7 +186,7 @@ define void @tc5_forced_ic2_vectorize_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -213,9 +213,9 @@ exit:
define void @tc5_forced_ic2_vectorize_i64(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_forced_ic2_vectorize_i64(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH1:.*:]]
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 2
@@ -228,11 +228,11 @@ define void @tc5_forced_ic2_vectorize_i64(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <2 x i64> [[TMP2]], ptr [[TMP3]], align 4
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
-; CHECK: [[SCALAR_PH]]:
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 4, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i64, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[VAL:%.*]] = load i64, ptr [[GEP_A]], align 4
@@ -240,7 +240,7 @@ define void @tc5_forced_ic2_vectorize_i64(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i64 [[ADD]], ptr [[GEP_B]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 5
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP6:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -321,9 +321,9 @@ exit:
define void @tc5_vf4_unsafe_useric_distance4_i32(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc5_vf4_unsafe_useric_distance4_i32(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*:]]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 4
@@ -335,11 +335,11 @@ define void @tc5_vf4_unsafe_useric_distance4_i32(ptr noalias %a, ptr noalias %b)
; CHECK-NEXT: store <4 x i32> [[TMP3]], ptr [[TMP0]], align 4
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 8, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 8, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[B_DST:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[B_SRC:%.*]] = getelementptr inbounds i32, ptr [[B_DST]], i64 -4
; CHECK-NEXT: [[DEP:%.*]] = load i32, ptr [[B_SRC]], align 4
@@ -349,7 +349,7 @@ define void @tc5_vf4_unsafe_useric_distance4_i32(ptr noalias %a, ptr noalias %b)
; CHECK-NEXT: store i32 [[ADD]], ptr [[B_DST]], align 4
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 9
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP7:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -416,15 +416,12 @@ exit:
ret void
}
-; FIXME: This is currently accepted as cost for vector is smaller than scalar.
-; Performance is poor due to poor CodeGen of using type promotion instead of
-; type widening.
-define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
-; CHECK-LABEL: define void @tc3_smin_i8_reject(
+define void @tc3_smin_i8_accept(ptr noalias %a, ptr noalias %b) #0 {
+; CHECK-LABEL: define void @tc3_smin_i8_accept(
; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH:.*:]]
-; CHECK-NEXT: br label %[[LOOP:.*]]
-; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*:]]
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i8>, ptr [[B]], align 1
@@ -433,11 +430,11 @@ define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store <2 x i8> [[TMP0]], ptr [[B]], align 1
; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
; CHECK: [[MIDDLE_BLOCK]]:
-; CHECK-NEXT: br label %[[SCALAR_PH1:.*]]
-; CHECK: [[SCALAR_PH1]]:
-; CHECK-NEXT: br label %[[LOOP1:.*]]
-; CHECK: [[LOOP1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 2, %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 2, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds i8, ptr [[B]], i64 [[IV]]
; CHECK-NEXT: [[TMP1:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
; CHECK-NEXT: [[ARRAYIDX2:%.*]] = getelementptr inbounds i8, ptr [[A]], i64 [[IV]]
@@ -446,7 +443,7 @@ define void @tc3_smin_i8_reject(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-NEXT: store i8 [[MIN]], ptr [[ARRAYIDX]], align 1
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i64 [[IV_NEXT]], 3
-; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP1]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP8:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
>From 25e64958a0e92e587ebdd4f16e2ffa213d2b5086 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Fri, 21 Aug 2026 15:32:54 +0100
Subject: [PATCH 27/32] Remove unneeded changes
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 10 ++++------
1 file changed, 4 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 6b183e3cfc1b7..caf4f064a594b 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3018,13 +3018,13 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
MaxPowerOf2RuntimeVF = std::nullopt; // Stick with tail-folding for now.
}
- auto NoScalarEpilogueNeeded = [this](unsigned MaxVF, unsigned EffectiveIC) {
+ auto NoScalarEpilogueNeeded = [this](unsigned MaxVF) {
// Return false if the loop is neither a single-latch-exit loop nor an
// early-exit loop as tail-folding is not supported in that case.
if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
!Legal->hasUncountableEarlyExit())
return false;
- unsigned MaxVFtimesIC = MaxVF * EffectiveIC;
+ unsigned MaxVFtimesIC = UserIC ? MaxVF * UserIC : MaxVF;;
ScalarEvolution *SE = PSE.getSE();
// Calling getSymbolicMaxBackedgeTakenCount enables support for loops
// with uncountable exits. For countable loops, the symbolic maximum must
@@ -3041,11 +3041,10 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
return Rem->isZero();
};
- unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
if (MaxPowerOf2RuntimeVF > 0u) {
assert((UserVF.isNonZero() || isPowerOf2_32(*MaxPowerOf2RuntimeVF)) &&
"MaxFixedVF must be a power of 2");
- if (NoScalarEpilogueNeeded(*MaxPowerOf2RuntimeVF, EffectiveIC)) {
+ if (NoScalarEpilogueNeeded(*MaxPowerOf2RuntimeVF)) {
// Accept MaxFixedVF if we do not have a tail.
LLVM_DEBUG(dbgs() << "LV: No tail will remain for any chosen VF.\n");
return MaxFactors;
@@ -3061,8 +3060,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// the trip count but the scalable factor does not, use the fixed-width
// factor in preference to allow the generation of a non-predicated loop.
if (EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop &&
- NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue(),
- EffectiveIC)) {
+ NoScalarEpilogueNeeded(MaxFactors.FixedVF.getFixedValue())) {
LLVM_DEBUG(dbgs() << "LV: Picking a fixed-width so that no tail will "
"remain for any chosen VF.\n");
MaxFactors.ScalableVF = ElementCount::getScalable(0);
>From b5bc46c234668a55af90cee9a81e041a1e1fd883 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Fri, 21 Aug 2026 15:33:32 +0100
Subject: [PATCH 28/32] Reintroduce UserIC
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index caf4f064a594b..e5e632249f345 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3018,7 +3018,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
MaxPowerOf2RuntimeVF = std::nullopt; // Stick with tail-folding for now.
}
- auto NoScalarEpilogueNeeded = [this](unsigned MaxVF) {
+ auto NoScalarEpilogueNeeded = [this, &UserIC](unsigned MaxVF) {
// Return false if the loop is neither a single-latch-exit loop nor an
// early-exit loop as tail-folding is not supported in that case.
if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
>From f9e4111a313eed96bae7f64cfc9b8877e02b333c Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Fri, 21 Aug 2026 15:34:22 +0100
Subject: [PATCH 29/32] format
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 9 +++++----
1 file changed, 5 insertions(+), 4 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index e5e632249f345..52c288528f22e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3024,7 +3024,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
if (TheLoop->getExitingBlock() != TheLoop->getLoopLatch() &&
!Legal->hasUncountableEarlyExit())
return false;
- unsigned MaxVFtimesIC = UserIC ? MaxVF * UserIC : MaxVF;;
+ unsigned MaxVFtimesIC = UserIC ? MaxVF * UserIC : MaxVF;
ScalarEvolution *SE = PSE.getSE();
// Calling getSymbolicMaxBackedgeTakenCount enables support for loops
// with uncountable exits. For countable loops, the symbolic maximum must
@@ -3067,7 +3067,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
return MaxFactors;
}
}
-
+
// Allow cases where the ExactTC == (VF * IC) + 1.
//
// This produces 1 vector iteration, and 1 scalar iteration with
@@ -5893,8 +5893,9 @@ LoopVectorizationPlanner::computeBestVF() {
if (ConsiderRegPressure)
RUs = calculateRegisterUsageForPlan(*P, VFs, TTI);
- if (!ForceVectorization && P->hasScalarTail() &&
- ExactTC.isFixed() && ExactTC.getFixedValue() > 0 && ExactTC.getFixedValue() <= TTI.getMinTripCountTailFoldingThreshold()) {
+ if (!ForceVectorization && P->hasScalarTail() && ExactTC.isFixed() &&
+ ExactTC.getFixedValue() > 0 &&
+ ExactTC.getFixedValue() <= TTI.getMinTripCountTailFoldingThreshold()) {
VFs = VFs.take_back(1);
}
>From cdd4a5a9717357be9107bd6f1a143b39a9dd51bb Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Mon, 24 Aug 2026 09:09:06 +0100
Subject: [PATCH 30/32] Update RISCV Tests
---
.../LoopVectorize/RISCV/low-trip-count.ll | 44 +++++++++++++++----
.../RISCV/select-invariant-cond-cost.ll | 20 +++++++--
.../LoopVectorize/RISCV/short-trip-count.ll | 36 +++++++--------
3 files changed, 69 insertions(+), 31 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count.ll b/llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count.ll
index c4e3f51eff4d6..9313260f998d1 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/low-trip-count.ll
@@ -44,18 +44,31 @@ define void @trip3_i8(ptr noalias nocapture noundef %dst, ptr noalias nocapture
; CHECK-LABEL: @trip3_i8(
; CHECK-NEXT: entry:
; CHECK-NEXT: br label [[FOR_BODY:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i8>, ptr [[DST:%.*]], align 1
+; CHECK-NEXT: [[TMP0:%.*]] = shl <2 x i8> [[WIDE_LOAD]], splat (i8 1)
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <2 x i8>, ptr [[DST1:%.*]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = add <2 x i8> [[TMP0]], [[WIDE_LOAD1]]
+; CHECK-NEXT: store <2 x i8> [[TMP1]], ptr [[DST1]], align 1
+; CHECK-NEXT: br label [[MIDDLE_BLOCK:%.*]]
+; CHECK: middle.block:
+; CHECK-NEXT: br label [[SCALAR_PH:%.*]]
+; CHECK: scalar.ph:
+; CHECK-NEXT: br label [[FOR_BODY1:%.*]]
; CHECK: for.body:
-; CHECK-NEXT: [[I_08:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INC:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds i8, ptr [[DST:%.*]], i64 [[I_08]]
+; CHECK-NEXT: [[I_08:%.*]] = phi i64 [ 2, [[SCALAR_PH]] ], [ [[INC:%.*]], [[FOR_BODY1]] ]
+; CHECK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 [[I_08]]
; CHECK-NEXT: [[TMP15:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
; CHECK-NEXT: [[MUL:%.*]] = shl i8 [[TMP15]], 1
-; CHECK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds i8, ptr [[DST1:%.*]], i64 [[I_08]]
+; CHECK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds i8, ptr [[DST1]], i64 [[I_08]]
; CHECK-NEXT: [[TMP16:%.*]] = load i8, ptr [[ARRAYIDX1]], align 1
; CHECK-NEXT: [[ADD:%.*]] = add i8 [[MUL]], [[TMP16]]
; CHECK-NEXT: store i8 [[ADD]], ptr [[ARRAYIDX1]], align 1
; CHECK-NEXT: [[INC]] = add nuw nsw i64 [[I_08]], 1
; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INC]], 3
-; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END:%.*]], label [[FOR_BODY]]
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END:%.*]], label [[FOR_BODY1]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: for.end:
; CHECK-NEXT: ret void
;
@@ -83,18 +96,31 @@ define void @trip5_i8(ptr noalias nocapture noundef %dst, ptr noalias nocapture
; CHECK-LABEL: @trip5_i8(
; CHECK-NEXT: entry:
; CHECK-NEXT: br label [[FOR_BODY:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[DST:%.*]], align 1
+; CHECK-NEXT: [[TMP0:%.*]] = shl <4 x i8> [[WIDE_LOAD]], splat (i8 1)
+; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[DST1:%.*]], align 1
+; CHECK-NEXT: [[TMP1:%.*]] = add <4 x i8> [[TMP0]], [[WIDE_LOAD1]]
+; CHECK-NEXT: store <4 x i8> [[TMP1]], ptr [[DST1]], align 1
+; CHECK-NEXT: br label [[MIDDLE_BLOCK:%.*]]
+; CHECK: middle.block:
+; CHECK-NEXT: br label [[SCALAR_PH:%.*]]
+; CHECK: scalar.ph:
+; CHECK-NEXT: br label [[FOR_BODY1:%.*]]
; CHECK: for.body:
-; CHECK-NEXT: [[I_08:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[INC:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds i8, ptr [[DST:%.*]], i64 [[I_08]]
+; CHECK-NEXT: [[I_08:%.*]] = phi i64 [ 4, [[SCALAR_PH]] ], [ [[INC:%.*]], [[FOR_BODY1]] ]
+; CHECK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 [[I_08]]
; CHECK-NEXT: [[TMP15:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
; CHECK-NEXT: [[MUL:%.*]] = shl i8 [[TMP15]], 1
-; CHECK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds i8, ptr [[DST1:%.*]], i64 [[I_08]]
+; CHECK-NEXT: [[ARRAYIDX1:%.*]] = getelementptr inbounds i8, ptr [[DST1]], i64 [[I_08]]
; CHECK-NEXT: [[TMP16:%.*]] = load i8, ptr [[ARRAYIDX1]], align 1
; CHECK-NEXT: [[ADD:%.*]] = add i8 [[MUL]], [[TMP16]]
; CHECK-NEXT: store i8 [[ADD]], ptr [[ARRAYIDX1]], align 1
; CHECK-NEXT: [[INC]] = add nuw nsw i64 [[I_08]], 1
; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INC]], 5
-; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END:%.*]], label [[FOR_BODY]]
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END:%.*]], label [[FOR_BODY1]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: for.end:
; CHECK-NEXT: ret void
;
@@ -360,7 +386,7 @@ define i8 @mul_non_pow_2_low_trip_count(ptr noalias %a) {
; CHECK-NEXT: [[MUL]] = mul i8 [[TMP5]], [[RDX]]
; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[IV_NEXT]], 10
-; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END:%.*]], label [[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label [[FOR_END:%.*]], label [[FOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: for.end:
; CHECK-NEXT: [[MUL_LCSSA:%.*]] = phi i8 [ [[MUL]], [[FOR_BODY]] ]
; CHECK-NEXT: ret i8 [[MUL_LCSSA]]
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/select-invariant-cond-cost.ll b/llvm/test/Transforms/LoopVectorize/RISCV/select-invariant-cond-cost.ll
index 8df8e0725e3fd..7df06ee6a0cef 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/select-invariant-cond-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/select-invariant-cond-cost.ll
@@ -8,20 +8,32 @@ target triple = "riscv64-unknown-linux-gnu"
define void @test_invariant_cond_for_select(ptr %dst, i8 %x) #0 {
; CHECK-LABEL: define void @test_invariant_cond_for_select(
; CHECK-SAME: ptr [[DST:%.*]], i8 [[X:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[C_1:%.*]] = icmp eq i8 [[X]], 0
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[TMP1:%.*]] = select i1 [[C_1]], <4 x i64> <i64 0, i64 1, i64 1, i64 1>, <4 x i64> zeroinitializer
+; CHECK-NEXT: [[TMP2:%.*]] = trunc <4 x i64> [[TMP1]] to <4 x i8>
+; CHECK-NEXT: call void @llvm.experimental.vp.strided.store.v4i8.p0.i64(<4 x i8> [[TMP2]], ptr align 1 [[DST]], i64 4, <4 x i1> splat (i1 true), i32 4)
+; CHECK-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP1:.*]]
+; CHECK: [[LOOP1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 16, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT: [[C_3:%.*]] = icmp eq i8 [[X]], 0
; CHECK-NEXT: [[C_2:%.*]] = icmp sgt i64 [[IV]], 0
; CHECK-NEXT: [[C_2_EXT:%.*]] = zext i1 [[C_2]] to i64
-; CHECK-NEXT: [[SEL:%.*]] = select i1 [[C_1]], i64 [[C_2_EXT]], i64 0
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[C_3]], i64 [[C_2_EXT]], i64 0
; CHECK-NEXT: [[SEL_TRUNC:%.*]] = trunc i64 [[SEL]] to i8
; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i8, ptr [[DST]], i64 [[IV]]
; CHECK-NEXT: store i8 [[SEL_TRUNC]], ptr [[GEP]], align 1
; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 4
; CHECK-NEXT: [[EC:%.*]] = icmp ult i64 [[IV]], 14
-; CHECK-NEXT: br i1 [[EC]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK-NEXT: br i1 [[EC]], label %[[LOOP1]], label %[[EXIT:.*]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/short-trip-count.ll b/llvm/test/Transforms/LoopVectorize/RISCV/short-trip-count.ll
index 74675437fae51..7521d28aed397 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/short-trip-count.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/short-trip-count.ll
@@ -5,15 +5,15 @@ define void @small_trip_count_min_vlen_128(ptr nocapture %a) vscale_range(4,1024
; CHECK-LABEL: @small_trip_count_min_vlen_128(
; CHECK-NEXT: entry:
; CHECK-NEXT: br label [[LOOP1:%.*]]
-; CHECK: loop:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], [[LOOP1]] ], [ 0, [[ENTRY:%.*]] ]
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[TMP1:%.*]], i32 [[IV]]
-; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[GEP]], align 4
-; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[V]], 1
-; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP]], align 4
-; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
-; CHECK-NEXT: [[COND:%.*]] = icmp eq i32 [[IV]], 3
-; CHECK-NEXT: br i1 [[COND]], label [[EXIT:%.*]], label [[LOOP1]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A:%.*]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[A]], align 4
+; CHECK-NEXT: br label [[MIDDLE_BLOCK:%.*]]
+; CHECK: middle.block:
+; CHECK-NEXT: br label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: ret void
;
@@ -38,15 +38,15 @@ define void @small_trip_count_min_vlen_32(ptr nocapture %a) vscale_range(1,1024)
; CHECK-LABEL: @small_trip_count_min_vlen_32(
; CHECK-NEXT: entry:
; CHECK-NEXT: br label [[LOOP1:%.*]]
-; CHECK: loop:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], [[LOOP1]] ], [ 0, [[ENTRY:%.*]] ]
-; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[TMP1:%.*]], i32 [[IV]]
-; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[GEP]], align 4
-; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[V]], 1
-; CHECK-NEXT: store i32 [[ADD]], ptr [[GEP]], align 4
-; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
-; CHECK-NEXT: [[COND:%.*]] = icmp eq i32 [[IV]], 3
-; CHECK-NEXT: br i1 [[COND]], label [[EXIT:%.*]], label [[LOOP1]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[A:%.*]], align 4
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], splat (i32 1)
+; CHECK-NEXT: store <4 x i32> [[TMP0]], ptr [[A]], align 4
+; CHECK-NEXT: br label [[MIDDLE_BLOCK:%.*]]
+; CHECK: middle.block:
+; CHECK-NEXT: br label [[EXIT:%.*]]
; CHECK: exit:
; CHECK-NEXT: ret void
;
>From c353fbb0e9cef69806933c21627011e3ed455248 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 26 Aug 2026 09:52:32 +0100
Subject: [PATCH 31/32] Respond to review comments
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 8 +++++---
1 file changed, 5 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 52c288528f22e..ee2ee1d4ee6b8 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3078,10 +3078,10 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// allow this form of transformation as this will increase CodeSize.
ElementCount ExactTC = getSmallConstantTripCount(PSE.getSE(), TheLoop);
unsigned TC = ExactTC.getFixedValue();
- unsigned MyMaxVF = 1ULL << Log2_32(TC);
+ unsigned MaxVFForTC = 1ULL << Log2_32(TC);
unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
- if (TC - MyMaxVF == 1 && !TheFunction->hasOptSize() && !Config.OptForSize) {
- unsigned VF = MyMaxVF / EffectiveIC;
+ if (TC - MaxVFForTC <= 1 && !TheFunction->hasOptSize() && !Config.OptForSize) {
+ unsigned VF = MaxVFForTC / EffectiveIC;
LLVM_DEBUG(dbgs() << "LV: Picking MaxVF=" << VF
<< " with 1 scalar iteration remaining.\n");
MaxFactors.FixedVF = ElementCount::getFixed(VF);
@@ -5893,6 +5893,8 @@ LoopVectorizationPlanner::computeBestVF() {
if (ConsiderRegPressure)
RUs = calculateRegisterUsageForPlan(*P, VFs, TTI);
+ // For loops where the Trip Count is below the Tail Folding Threshold, only consider the largest VF to ensure, where TC == VF * IC or TC - 1 == VF * IC, one vector iteration, and one scalar iteration if needed is generated.
+ // FIXME: Encode this directly in LVPlanner rather than as part of the LoopVectorizer.
if (!ForceVectorization && P->hasScalarTail() && ExactTC.isFixed() &&
ExactTC.getFixedValue() > 0 &&
ExactTC.getFixedValue() <= TTI.getMinTripCountTailFoldingThreshold()) {
>From b6c3e76b135fbff82bac2f1c1f37e8e0816e0b21 Mon Sep 17 00:00:00 2001
From: Jack Styles <jack.styles at arm.com>
Date: Wed, 26 Aug 2026 11:44:44 +0100
Subject: [PATCH 32/32] format
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 11 ++++++++---
1 file changed, 8 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index ee2ee1d4ee6b8..55f3e7413ce04 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3080,7 +3080,8 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
unsigned TC = ExactTC.getFixedValue();
unsigned MaxVFForTC = 1ULL << Log2_32(TC);
unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
- if (TC - MaxVFForTC <= 1 && !TheFunction->hasOptSize() && !Config.OptForSize) {
+ if (TC - MaxVFForTC <= 1 && !TheFunction->hasOptSize() &&
+ !Config.OptForSize) {
unsigned VF = MaxVFForTC / EffectiveIC;
LLVM_DEBUG(dbgs() << "LV: Picking MaxVF=" << VF
<< " with 1 scalar iteration remaining.\n");
@@ -5893,8 +5894,12 @@ LoopVectorizationPlanner::computeBestVF() {
if (ConsiderRegPressure)
RUs = calculateRegisterUsageForPlan(*P, VFs, TTI);
- // For loops where the Trip Count is below the Tail Folding Threshold, only consider the largest VF to ensure, where TC == VF * IC or TC - 1 == VF * IC, one vector iteration, and one scalar iteration if needed is generated.
- // FIXME: Encode this directly in LVPlanner rather than as part of the LoopVectorizer.
+ // For loops where the Trip Count is below the Tail Folding Threshold, only
+ // consider the largest VF to ensure, where TC == VF * IC or TC - 1 == VF *
+ // IC, one vector iteration, and one scalar iteration if needed is
+ // generated.
+ // FIXME: Encode this directly in LVPlanner rather than as part of the
+ // LoopVectorizer.
if (!ForceVectorization && P->hasScalarTail() && ExactTC.isFixed() &&
ExactTC.getFixedValue() > 0 &&
ExactTC.getFixedValue() <= TTI.getMinTripCountTailFoldingThreshold()) {
More information about the llvm-commits
mailing list