[llvm] 899d817 - [LV] Skip low-trip count logic there is no scalar tail. (#225633)
via llvm-commits
llvm-commits at lists.llvm.org
Sat Sep 26 13:13:43 PDT 2026
Author: Florian Hahn
Date: 2026-09-26T20:13:37Z
New Revision: 899d817c7950c997402cd229935cd822acf45b08
URL: https://github.com/llvm/llvm-project/commit/899d817c7950c997402cd229935cd822acf45b08
DIFF: https://github.com/llvm/llvm-project/commit/899d817c7950c997402cd229935cd822acf45b08.diff
LOG: [LV] Skip low-trip count logic there is no scalar tail. (#225633)
https://github.com/llvm/llvm-project/pull/195823 added logic consider
vectorization of low trip count loops if there was no or a single
iteration remaining.
This causes loops to be vectorized with a VF where no scalar tail
remains, even if it is required for legality (loop with multiple
countable exits require scalar epilogue to pick the exit).
For now, limit to cases where there's a scalar iteration remaining.
PR: https://github.com/llvm/llvm-project/pull/225633
Added:
Modified:
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll
llvm/test/Transforms/LoopVectorize/RISCV/countable-early-exit-no-epilogue.ll
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 73588103e26d7..e934fec366331 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3067,11 +3067,11 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
return MaxFactors;
}
- // Allow cases where the ExactTC == (VF * IC) or ExactTC == (VF * IC) + 1.
+ // Allow cases where the ExactTC == (VF * IC) + 1.
//
- // This produces at most 1 vector iteration, and at most 1 scalar iteration
- // with no remainder. Later passes will eliminate the loop and leave
- // straight-line code as the both iteration counts are statically known.
+ // This produces 1 vector iteration, and 1 scalar iteration with no
+ // remainder. Later passes will eliminate the loop and leave straight-line
+ // code as the both iteration counts are statically known.
//
// If a function is marked as minsize/optsize or OptForSize is set, do not
// allow this form of transformation as this will increase CodeSize.
@@ -3080,7 +3080,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// enough to accurately determine if vectorization is beneficial.
unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
unsigned MaxVFForTC = llvm::bit_floor(TC.getFixedValue());
- if (TC.getFixedValue() - MaxVFForTC <= 1 && MaxVFForTC / EffectiveIC > 1 &&
+ if (TC.getFixedValue() - MaxVFForTC == 1 && MaxVFForTC / EffectiveIC > 1 &&
MaxVFForTC <= (MaxFactors.FixedVF.getFixedValue() * EffectiveIC) &&
!Config.OptForSize) {
unsigned NumOfInstructions = llvm::sum_of(
@@ -3090,7 +3090,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
if (NumOfInstructions > LowTripCountLoopBodySizeLimit) {
unsigned VF = MaxVFForTC / EffectiveIC;
LLVM_DEBUG(dbgs() << "LV: Picking MaxVF=" << VF
- << " with at most 1 scalar iteration remaining.\n");
+ << " with 1 scalar iteration remaining.\n");
MaxFactors.FixedVF = ElementCount::getFixed(VF);
MaxFactors.ScalableVF = ElementCount::getScalable(0);
return MaxFactors;
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll b/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll
index be740c8f0fa16..24d94e5354bda 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/sve-low-trip-count.ll
@@ -617,7 +617,7 @@ exit:
define void @tc4_vf4(ptr noalias %a, ptr noalias %b) #0 {
; CHECK-LABEL: define void @tc4_vf4(
-; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-SAME: ptr noalias [[A:%.*]], ptr noalias [[B:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
@@ -690,6 +690,48 @@ exit:
ret void
}
+define i32 @countable_early_exit_tc_4(ptr noalias %b) #0 {
+; CHECK-LABEL: define i32 @countable_early_exit_tc_4(
+; CHECK-SAME: ptr noalias [[B:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT: [[C:%.*]] = icmp eq i64 [[IV]], 3
+; CHECK-NEXT: br i1 [[C]], label %[[EXIT1:.*]], label %[[LATCH]]
+; CHECK: [[LATCH]]:
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; CHECK-NEXT: store i32 1, ptr [[GEP]], align 4
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 100
+; CHECK-NEXT: br i1 [[EC]], label %[[EXIT2:.*]], label %[[LOOP]]
+; CHECK: [[EXIT1]]:
+; CHECK-NEXT: ret i32 1
+; CHECK: [[EXIT2]]:
+; CHECK-NEXT: ret i32 2
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+ %c = icmp eq i64 %iv, 3
+ br i1 %c, label %exit1, label %latch
+
+latch:
+ %gep = getelementptr inbounds i32, ptr %b, i64 %iv
+ store i32 1, ptr %gep, align 4
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 100
+ br i1 %ec, label %exit2, label %loop
+
+exit1:
+ ret i32 1
+
+exit2:
+ ret i32 2
+}
+
attributes #0 = { vscale_range(1,16) "target-features"="+sve" }
attributes #1 = { vscale_range(1,16) "target-features"="+sve" minsize }
attributes #2 = { vscale_range(1,16) "target-features"="+sve" optsize }
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/countable-early-exit-no-epilogue.ll b/llvm/test/Transforms/LoopVectorize/RISCV/countable-early-exit-no-epilogue.ll
index 01cc2c2f7cff0..c1e2913e5be38 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/countable-early-exit-no-epilogue.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/countable-early-exit-no-epilogue.ll
@@ -3,20 +3,22 @@
; RUN: opt -passes=loop-vectorize -mtriple=riscv64 -mattr=+v -low-trip-count-loop-body-size-limit=0 -tail-folding-policy=must-fold-tail -S %s | FileCheck %s --check-prefix=NO-EPILOGUE
; RUN: opt -passes=loop-vectorize -mtriple=riscv64 -mattr=+v -vectorizer-min-trip-count=0 -tail-folding-policy=dont-fold-tail -force-vector-width=2 -S %s | FileCheck %s --check-prefix=EPILOGUE
-; FIXME: Currently this gets miscompiled, as the forced scalar epilogue is missing.
define i32 @countable_early_exit(ptr noalias %b) {
; NO-EPILOGUE-LABEL: define i32 @countable_early_exit(
; NO-EPILOGUE-SAME: ptr noalias [[B:%.*]]) #[[ATTR0:[0-9]+]] {
-; NO-EPILOGUE-NEXT: [[ENTRY:.*:]]
-; NO-EPILOGUE-NEXT: br label %[[VECTOR_PH:.*]]
-; NO-EPILOGUE: [[VECTOR_PH]]:
-; NO-EPILOGUE-NEXT: br label %[[VECTOR_BODY:.*]]
-; NO-EPILOGUE: [[VECTOR_BODY]]:
-; NO-EPILOGUE-NEXT: store <4 x i32> splat (i32 1), ptr [[B]], align 4
-; NO-EPILOGUE-NEXT: br label %[[MIDDLE_BLOCK:.*]]
-; NO-EPILOGUE: [[MIDDLE_BLOCK]]:
-; NO-EPILOGUE-NEXT: br label %[[EXIT2:.*]]
-; NO-EPILOGUE: [[EXIT1:.*:]]
+; NO-EPILOGUE-NEXT: [[ENTRY:.*]]:
+; NO-EPILOGUE-NEXT: br label %[[LOOP:.*]]
+; NO-EPILOGUE: [[LOOP]]:
+; NO-EPILOGUE-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; NO-EPILOGUE-NEXT: [[C:%.*]] = icmp eq i64 [[IV]], 3
+; NO-EPILOGUE-NEXT: br i1 [[C]], label %[[EXIT1:.*]], label %[[LATCH]]
+; NO-EPILOGUE: [[LATCH]]:
+; NO-EPILOGUE-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; NO-EPILOGUE-NEXT: store i32 1, ptr [[GEP]], align 4
+; NO-EPILOGUE-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; NO-EPILOGUE-NEXT: [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 100
+; NO-EPILOGUE-NEXT: br i1 [[EC]], label %[[EXIT2:.*]], label %[[LOOP]]
+; NO-EPILOGUE: [[EXIT1]]:
; NO-EPILOGUE-NEXT: ret i32 1
; NO-EPILOGUE: [[EXIT2]]:
; NO-EPILOGUE-NEXT: ret i32 2
More information about the llvm-commits
mailing list