[llvm] 58b1026 - [SCEV] - Add positive-stride predicate for backedge-taken count. (#222261)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 29 01:15:17 PDT 2026
Author: Pawan Nirpal
Date: 2026-09-29T13:45:11+05:30
New Revision: 58b102609a9d08fbadb9c3fa1dd88bba2cf7f711
URL: https://github.com/llvm/llvm-project/commit/58b102609a9d08fbadb9c3fa1dd88bba2cf7f711
DIFF: https://github.com/llvm/llvm-project/commit/58b102609a9d08fbadb9c3fa1dd88bba2cf7f711.diff
LOG: [SCEV] - Add positive-stride predicate for backedge-taken count. (#222261)
When `howManyLessThans` encounters a loop with an unknown stride that
cannot be proven finite (no `mustprogress` or side-effect-free
guarantee), SCEV currently returns `CouldNotCompute` for the
backedge-taken count. This blocks downstream consumers like the loop
vectorizer from optimizing such loops.
This patch relaxes the requirement by allowing a predicated
backedge-taken count when `AllowPredicates` is true. Instead of
requiring `loopIsFiniteByAssumption(L)` unconditionally, we add a
`Compare predicate: stride sgt 0` when finiteness cannot be proven.
A positive stride guarantees forward progress, making the BTC formula
correct. The predicate is emitted as a runtime check by consumers
(e.g., the loop vectorizer generates a guard branch before the vector
loop).
This is the SCEV-level fix for a class of variable-increment loops
common in Fortran-style benchmarks, e.g.:
void f(int n, int *a, int *b, int *c, int inc) {
int i = 0;
L10: if (i >= n) goto L20;
a[i] = a[i] + b[i] * c[i];
i = i + inc; // inc unknown at compile time
goto L10;
L20: ;
}
Addressing : https://github.com/llvm/llvm-project/issues/221915
---------
Co-authored-by: Florian Hahn <flo at fhahn.com>
Added:
llvm/test/Analysis/ScalarEvolution/trip-count-variable-stride-predicate.ll
llvm/test/Transforms/LoopVectorize/scev-variable-stride-predicate.ll
Modified:
llvm/lib/Analysis/ScalarEvolution.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Analysis/ScalarEvolution.cpp b/llvm/lib/Analysis/ScalarEvolution.cpp
index 6a244d93a647c..cbadeaf7b347d 100644
--- a/llvm/lib/Analysis/ScalarEvolution.cpp
+++ b/llvm/lib/Analysis/ScalarEvolution.cpp
@@ -13513,7 +13513,9 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
// a) IV is either nuw or nsw depending upon signedness (indicated by the
// NoWrap flag).
// b) the loop is guaranteed to be finite (e.g. is mustprogress and has
- // no side effects within the loop)
+ // b) the loop is guaranteed to be finite (e.g. is mustprogress and has
+ // no side effects within the loop) or a predicate is added to ensure
+ // stride is positive.
// c) loop has a single static exit (with no abnormal exits)
//
// Precondition a) implies that if the stride is negative, this is a single
@@ -13526,11 +13528,23 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
// The positive stride case is the same as isKnownPositive(Stride) returning
// true (original behavior of the function).
//
- if (PredicatedIV || !NoWrap || !loopIsFiniteByAssumption(L) ||
- !loopHasNoAbnormalExits(L))
+ if (PredicatedIV || !NoWrap || !loopHasNoAbnormalExits(L))
return getCouldNotCompute();
- if (!isKnownNonZero(Stride)) {
+ if (!loopIsFiniteByAssumption(L)) {
+ // If the loop may be infinite, add a predicate ensuring Stride is
+ // positive, to guarantee forward progress.
+ if (!AllowPredicates || !isLoopInvariant(Stride, L))
+ return getCouldNotCompute();
+
+ const SCEV *Zero = getZero(Stride->getType());
+ const SCEVPredicate *P =
+ getComparePredicate(ICmpInst::ICMP_SGT, Stride, Zero);
+ Predicates.push_back(P);
+ // When the predicate holds (Stride > 0), umax(Stride, 1) == Stride,
+ // so the result is unchanged. To prevent div by zero.
+ Stride = getUMaxExpr(Stride, getOne(Stride->getType()));
+ } else if (!isKnownNonZero(Stride)) {
// If we have a step of zero, and RHS isn't invariant in L, we don't know
// if it might eventually be greater than start and if so, on which
// iteration. We can't even produce a useful upper bound.
diff --git a/llvm/test/Analysis/ScalarEvolution/trip-count-variable-stride-predicate.ll b/llvm/test/Analysis/ScalarEvolution/trip-count-variable-stride-predicate.ll
new file mode 100644
index 0000000000000..42a91320182de
--- /dev/null
+++ b/llvm/test/Analysis/ScalarEvolution/trip-count-variable-stride-predicate.ll
@@ -0,0 +1,31 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -disable-output "-passes=print<scalar-evolution>" -scalar-evolution-classify-expressions=0 2>&1 | FileCheck %s
+
+define void @variable_stride_no_mustprogress(i64 %n, i64 %stride) {
+; CHECK-LABEL: 'variable_stride_no_mustprogress'
+; CHECK-NEXT: Determining loop execution counts for: @variable_stride_no_mustprogress
+; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Predicated backedge-taken count is ((((-1 * (1 umin ((-1 * %stride) + (%n smax %stride))))<nuw><nsw> + (-1 * %stride) + (%n smax %stride)) /u (1 umax %stride)) + (1 umin ((-1 * %stride) + (%n smax %stride))))
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Compare predicate: %stride sgt) 0
+; CHECK-NEXT: Loop %loop: Predicated constant max backedge-taken count is i64 -1
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Compare predicate: %stride sgt) 0
+; CHECK-NEXT: Loop %loop: Predicated symbolic max backedge-taken count is ((((-1 * (1 umin ((-1 * %stride) + (%n smax %stride))))<nuw><nsw> + (-1 * %stride) + (%n smax %stride)) /u (1 umax %stride)) + (1 umin ((-1 * %stride) + (%n smax %stride))))
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Compare predicate: %stride sgt) 0
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %iv.next = add nsw i64 %iv, %stride
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/scev-variable-stride-predicate.ll b/llvm/test/Transforms/LoopVectorize/scev-variable-stride-predicate.ll
new file mode 100644
index 0000000000000..2d532aacc8070
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/scev-variable-stride-predicate.ll
@@ -0,0 +1,102 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "scalar.ph:" --version 6
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 -force-vector-interleave=1 -S %s | FileCheck %s
+
+define void @variable_stride_predicate(i64 %n, ptr noalias %a, ptr noalias %b, i64 %stride) {
+; CHECK-LABEL: define void @variable_stride_predicate(
+; CHECK-SAME: i64 [[N:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], i64 [[STRIDE:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK: [[VECTOR_SCEVCHECK]]:
+; CHECK-NEXT: [[IDENT_CHECK2:%.*]] = icmp ne i64 [[STRIDE]], 1
+; CHECK-NEXT: br i1 [[IDENT_CHECK2]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP1:%.*]] = and i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i32, ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD4:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD4]]
+; CHECK-NEXT: store <4 x i32> [[TMP4]], ptr [[TMP2]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP5]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP0]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %gep.a = getelementptr inbounds i32, ptr %a, i64 %iv
+ %load.a = load i32, ptr %gep.a, align 4
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %iv
+ %load.b = load i32, ptr %gep.b, align 4
+ %add = add nsw i32 %load.a, %load.b
+ store i32 %add, ptr %gep.a, align 4
+ %iv.next = add nsw i64 %iv, %stride
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+define void @variable_stride_separate_iv(ptr %a, i64 %n, i64 %stride) {
+; CHECK-LABEL: define void @variable_stride_separate_iv(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i64 [[STRIDE:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.smax.i64(i64 [[STRIDE]], i64 [[N]])
+; CHECK-NEXT: [[TMP1:%.*]] = sub i64 [[TMP0]], [[STRIDE]]
+; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP1]], i64 1)
+; CHECK-NEXT: [[TMP3:%.*]] = sub i64 [[TMP1]], [[TMP2]]
+; CHECK-NEXT: [[TMP4:%.*]] = call i64 @llvm.umax.i64(i64 [[STRIDE]], i64 1)
+; CHECK-NEXT: [[TMP5:%.*]] = udiv i64 [[TMP3]], [[TMP4]]
+; CHECK-NEXT: [[TMP6:%.*]] = add i64 [[TMP2]], [[TMP5]]
+; CHECK-NEXT: [[TMP7:%.*]] = add i64 [[TMP6]], 1
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP7]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK: [[VECTOR_SCEVCHECK]]:
+; CHECK-NEXT: [[IDENT_CHECK:%.*]] = icmp sle i64 [[STRIDE]], 0
+; CHECK-NEXT: br i1 [[IDENT_CHECK]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP8:%.*]] = and i64 [[TMP7]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP7]], [[TMP8]]
+; CHECK-NEXT: [[TMP9:%.*]] = mul i64 [[N_VEC]], [[STRIDE]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> splat (i32 7), ptr [[TMP10]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP11]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[TMP7]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], [[EXIT:label %.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %j = phi i64 [ 0, %entry ], [ %j.next, %loop ]
+ %gep = getelementptr inbounds i32, ptr %a, i64 %j
+ store i32 7, ptr %gep, align 4
+ %j.next = add nuw nsw i64 %j, 1
+ %iv.next = add nsw i64 %iv, %stride
+ %cmp = icmp slt i64 %iv.next, %n
+ br i1 %cmp, label %loop, label %exit
+
+exit:
+ ret void
+}
More information about the llvm-commits
mailing list