[llvm] [LV] Skip low-trip count logic there is no scalar tail (PR #225633)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 23 01:08:17 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: Florian Hahn (fhahn)
<details>
<summary>Changes</summary>
https://github.com/llvm/llvm-project/pull/195823 added logic consider vectorization of low trip count loops if there was no or a single iteration remaining.
This causes loops to be vectorized with a VF where no scalar tail remains, even if it is required for legality (loop with multiple countable exits require scalar epilogue to pick the exit).
For now, limit to cases where there's a scalar iteration remaining.
---
Full diff: https://github.com/llvm/llvm-project/pull/225633.diff
2 Files Affected:
- (modified) llvm/lib/Transforms/Vectorize/LoopVectorize.cpp (+5-5)
- (modified) llvm/test/Transforms/LoopVectorize/RISCV/countable-early-exit-no-epilogue.ll (+10-8)
``````````diff
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 805a57f8dc4ab..b0f3adf203554 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -3075,11 +3075,11 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
}
}
- // Allow cases where the ExactTC == (VF * IC) or ExactTC == (VF * IC) + 1.
+ // Allow cases where the ExactTC == (VF * IC) + 1.
//
- // This produces at most 1 vector iteration, and at most 1 scalar iteration
- // with no remainder. Later passes will eliminate the loop and leave
- // straight-line code as the both iteration counts are statically known.
+ // This produces t 1 vector iteration, and 1 scalar iteration with no
+ // remainder. Later passes will eliminate the loop and leave straight-line
+ // code as the both iteration counts are statically known.
//
// If a function is marked as minsize/optsize or OptForSize is set, do not
// allow this form of transformation as this will increase CodeSize.
@@ -3088,7 +3088,7 @@ LoopVectorizationCostModel::computeMaxVF(ElementCount UserVF, unsigned UserIC) {
// enough to accurately determine if vectorization is beneficial.
unsigned EffectiveIC = UserIC > 0 ? UserIC : 1;
unsigned MaxVFForTC = llvm::bit_floor(TC.getFixedValue());
- if (TC.getFixedValue() - MaxVFForTC <= 1 && MaxVFForTC / EffectiveIC > 1 &&
+ if (TC.getFixedValue() - MaxVFForTC == 1 && MaxVFForTC / EffectiveIC > 1 &&
MaxVFForTC <= (MaxFactors.FixedVF.getFixedValue() * EffectiveIC) &&
!Config.OptForSize) {
unsigned NumOfInstructions = llvm::sum_of(
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/countable-early-exit-no-epilogue.ll b/llvm/test/Transforms/LoopVectorize/RISCV/countable-early-exit-no-epilogue.ll
index 349a026cc1b4a..5fe6767d814b4 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/countable-early-exit-no-epilogue.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/countable-early-exit-no-epilogue.ll
@@ -3,20 +3,22 @@
; RUN: opt -passes=loop-vectorize -mtriple=riscv64 -mattr=+v -low-trip-count-loop-body-size-limit=0 -tail-folding-policy=must-fold-tail -S %s | FileCheck %s --check-prefix=NO-EPILOGUE
; RUN: opt -passes=loop-vectorize -mtriple=riscv64 -mattr=+v -vectorizer-min-trip-count=0 -tail-folding-policy=dont-fold-tail -force-vector-width=2 -S %s | FileCheck %s --check-prefix=EPILOGUE
-; FIXME: Currently this gets miscompiled, as the forced scalar epilogue is missing.
define i32 @countable_early_exit(ptr noalias %b) {
; NO-EPILOGUE-LABEL: define i32 @countable_early_exit(
; NO-EPILOGUE-SAME: ptr noalias [[B:%.*]]) #[[ATTR0:[0-9]+]] {
-; NO-EPILOGUE-NEXT: [[ENTRY:.*:]]
-; NO-EPILOGUE-NEXT: br label %[[VECTOR_PH:.*]]
-; NO-EPILOGUE: [[VECTOR_PH]]:
+; NO-EPILOGUE-NEXT: [[VECTOR_PH:.*]]:
; NO-EPILOGUE-NEXT: br label %[[VECTOR_BODY:.*]]
; NO-EPILOGUE: [[VECTOR_BODY]]:
-; NO-EPILOGUE-NEXT: store <4 x i32> splat (i32 1), ptr [[B]], align 4
-; NO-EPILOGUE-NEXT: br label %[[MIDDLE_BLOCK:.*]]
+; NO-EPILOGUE-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[IV_NEXT:%.*]], %[[MIDDLE_BLOCK:.*]] ]
+; NO-EPILOGUE-NEXT: [[C:%.*]] = icmp eq i64 [[IV]], 3
+; NO-EPILOGUE-NEXT: br i1 [[C]], label %[[EXIT1:.*]], label %[[MIDDLE_BLOCK]]
; NO-EPILOGUE: [[MIDDLE_BLOCK]]:
-; NO-EPILOGUE-NEXT: br label %[[EXIT2:.*]]
-; NO-EPILOGUE: [[EXIT1:.*:]]
+; NO-EPILOGUE-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[IV]]
+; NO-EPILOGUE-NEXT: store i32 1, ptr [[GEP]], align 4
+; NO-EPILOGUE-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; NO-EPILOGUE-NEXT: [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], 100
+; NO-EPILOGUE-NEXT: br i1 [[EC]], label %[[EXIT2:.*]], label %[[VECTOR_BODY]]
+; NO-EPILOGUE: [[EXIT1]]:
; NO-EPILOGUE-NEXT: ret i32 1
; NO-EPILOGUE: [[EXIT2]]:
; NO-EPILOGUE-NEXT: ret i32 2
``````````
</details>
https://github.com/llvm/llvm-project/pull/225633
More information about the llvm-commits
mailing list