[llvm] [SCEV] - Add positive-stride predicate for backedge-taken count. (PR #222261)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 01:03:09 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: Pawan Nirpal (pawan-nirpal-031)
<details>
<summary>Changes</summary>
When `howManyLessThans` encounters a loop with an unknown stride that
cannot be proven finite (no `mustprogress` or side-effect-free
guarantee), SCEV currently returns `CouldNotCompute` for the
backedge-taken count. This blocks downstream consumers like the loop
vectorizer from optimizing such loops.
This patch relaxes the requirement by allowing a predicated
backedge-taken count when `AllowPredicates` is true. Instead of
requiring `loopIsFiniteByAssumption(L)` unconditionally, we add a
`Compare predicate: stride sgt 0` when finiteness cannot be proven.
A positive stride guarantees forward progress, making the BTC formula
correct. The predicate is emitted as a runtime check by consumers
(e.g., the loop vectorizer generates a guard branch before the vector
loop).
This is the SCEV-level fix for a class of variable-increment loops
common in Fortran-style benchmarks, e.g.:
void f(int n, int *a, int *b, int *c, int inc) {
int i = 0;
L10: if (i >= n) goto L20;
a[i] = a[i] + b[i] * c[i];
i = i + inc; // inc unknown at compile time
goto L10;
L20: ;
}
Addressing : https://github.com/llvm/llvm-project/issues/221915
---
Full diff: https://github.com/llvm/llvm-project/pull/222261.diff
3 Files Affected:
- (modified) llvm/lib/Analysis/ScalarEvolution.cpp (+14-3)
- (added) llvm/test/Analysis/ScalarEvolution/trip-count-variable-stride-predicate.ll (+53)
- (added) llvm/test/Transforms/LoopVectorize/RISCV/scev-variable-stride-predicate.ll (+75)
``````````diff
diff --git a/llvm/lib/Analysis/ScalarEvolution.cpp b/llvm/lib/Analysis/ScalarEvolution.cpp
index 9211b3d60b6ed..0fd5033a50e46 100644
--- a/llvm/lib/Analysis/ScalarEvolution.cpp
+++ b/llvm/lib/Analysis/ScalarEvolution.cpp
@@ -13467,11 +13467,22 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
// The positive stride case is the same as isKnownPositive(Stride) returning
// true (original behavior of the function).
//
- if (PredicatedIV || !NoWrap || !loopIsFiniteByAssumption(L) ||
- !loopHasNoAbnormalExits(L))
+ if (PredicatedIV || !NoWrap || !loopHasNoAbnormalExits(L))
return getCouldNotCompute();
- if (!isKnownNonZero(Stride)) {
+ if (!loopIsFiniteByAssumption(L)) {
+ // If we cannot prove the loop is finite but predicates are allowed,
+ // we can add a predicate that the stride is positive. This ensures
+ // the loop makes forward progress and the BTC formula is correct.
+ // The predicate will be emitted as a runtime check by the consumer
+ // (e.g., the loop vectorizer), guarding the optimized loop version.
+ if (!AllowPredicates || !isLoopInvariant(Stride, L))
+ return getCouldNotCompute();
+
+ const SCEV *Zero = getZero(Stride->getType());
+ auto *P = getComparePredicate(ICmpInst::ICMP_SGT, Stride, Zero);
+ Predicates.push_back(P);
+ } else if (!isKnownNonZero(Stride)) {
// If we have a step of zero, and RHS isn't invariant in L, we don't know
// if it might eventually be greater than start and if so, on which
// iteration. We can't even produce a useful upper bound.
diff --git a/llvm/test/Analysis/ScalarEvolution/trip-count-variable-stride-predicate.ll b/llvm/test/Analysis/ScalarEvolution/trip-count-variable-stride-predicate.ll
new file mode 100644
index 0000000000000..9a986261de40f
--- /dev/null
+++ b/llvm/test/Analysis/ScalarEvolution/trip-count-variable-stride-predicate.ll
@@ -0,0 +1,53 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 4
+; RUN: opt < %s -disable-output "-passes=print<scalar-evolution>" -scalar-evolution-classify-expressions=0 2>&1 | FileCheck %s
+
+define void @variable_stride_predicated_scev(i32 %n, ptr noalias %a, ptr noalias %b, ptr noalias %c, i32 %inc) #0 {
+; CHECK-LABEL: 'variable_stride_predicated_scev'
+; CHECK-NEXT: Determining loop execution counts for: @variable_stride_predicated_scev
+; CHECK-NEXT: Loop %if.end: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %if.end: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %if.end: Unpredictable symbolic max backedge-taken count.
+; CHECK-NEXT: Loop %if.end: Predicated backedge-taken count is ((-1 + ((zext i32 %n to i64) smax (sext i32 %inc to i64)))<nsw> /u (sext i32 %inc to i64))
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Compare predicate: (sext i32 %inc to i64) sgt) 0
+; CHECK-NEXT: Loop %if.end: Predicated constant max backedge-taken count is i64 6442450943
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Compare predicate: (sext i32 %inc to i64) sgt) 0
+; CHECK-NEXT: Loop %if.end: Predicated symbolic max backedge-taken count is ((-1 + ((zext i32 %n to i64) smax (sext i32 %inc to i64)))<nsw> /u (sext i32 %inc to i64))
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Compare predicate: (sext i32 %inc to i64) sgt) 0
+;
+entry:
+ %cmp.not14 = icmp sgt i32 %n, 0
+ br i1 %cmp.not14, label %if.end.preheader, label %L20
+
+if.end.preheader: ; preds = %entry
+ %0 = sext i32 %inc to i64
+ %1 = zext nneg i32 %n to i64
+ br label %if.end
+
+if.end: ; preds = %if.end.preheader, %if.end
+ %indvars.iv = phi i64 [ 0, %if.end.preheader ], [ %indvars.iv.next, %if.end ]
+ %arrayidx = getelementptr inbounds [4 x i8], ptr %a, i64 %indvars.iv
+ %2 = load i32, ptr %arrayidx, align 4, !tbaa !13
+ %arrayidx2 = getelementptr inbounds [4 x i8], ptr %b, i64 %indvars.iv
+ %3 = load i32, ptr %arrayidx2, align 4, !tbaa !13
+ %arrayidx4 = getelementptr inbounds [4 x i8], ptr %c, i64 %indvars.iv
+ %4 = load i32, ptr %arrayidx4, align 4, !tbaa !13
+ %mul = mul nsw i32 %4, %3
+ %add = add nsw i32 %mul, %2
+ store i32 %add, ptr %arrayidx, align 4, !tbaa !13
+ %indvars.iv.next = add nsw i64 %indvars.iv, %0
+ %cmp.not = icmp slt i64 %indvars.iv.next, %1
+ br i1 %cmp.not, label %if.end, label %L20
+
+L20: ; preds = %if.end, %entry
+ ret void
+}
+
+attributes #0 = { nofree norecurse nosync nounwind memory(argmem: readwrite) uwtable vscale_range(4,1024) "no-trapping-math"="true" "stack-protector-buffer-size"="8" "target-cpu"="spacemit-x60" "target-features"="+64bit,+v" }
+
+!10 = !{!"int", !11, i64 0}
+!11 = !{!"omnipotent char", !12, i64 0}
+!12 = !{!"Simple C/C++ TBAA"}
+!13 = !{!10, !10, i64 0}
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/scev-variable-stride-predicate.ll b/llvm/test/Transforms/LoopVectorize/RISCV/scev-variable-stride-predicate.ll
new file mode 100644
index 0000000000000..e68784e3388f3
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/scev-variable-stride-predicate.ll
@@ -0,0 +1,75 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "scalar.ph:" --version 5
+; RUN: opt -passes=loop-vectorize -mtriple=riscv64 -S %s | FileCheck %s
+
+define void @variable_stride_predicated_vectorize(i32 %n, ptr noalias %a, ptr noalias %b, ptr noalias %c, i32 %inc) #0 {
+; CHECK-LABEL: define void @variable_stride_predicated_vectorize(
+; CHECK-SAME: i32 [[N:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]], i32 [[INC:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[CMP_NOT14:%.*]] = icmp sgt i32 [[N]], 0
+; CHECK-NEXT: br i1 [[CMP_NOT14]], label %[[IF_END_PREHEADER:.*]], [[L20:label %.*]]
+; CHECK: [[IF_END_PREHEADER]]:
+; CHECK-NEXT: [[TMP0:%.*]] = sext i32 [[INC]] to i64
+; CHECK-NEXT: [[TMP1:%.*]] = zext nneg i32 [[N]] to i64
+; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.smax.i64(i64 [[TMP1]], i64 1)
+; CHECK-NEXT: br label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK: [[VECTOR_SCEVCHECK]]:
+; CHECK-NEXT: [[IDENT_CHECK2:%.*]] = icmp ne i32 [[INC]], 1
+; CHECK-NEXT: br i1 [[IDENT_CHECK2]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[CURRENT_ITERATION_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[AVL:%.*]] = phi i64 [ [[TMP2]], %[[VECTOR_PH]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 4, i1 true)
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds [4 x i8], ptr [[A]], i64 [[INDEX]]
+; CHECK-NEXT: [[VP_OP_LOAD:%.*]] = call <vscale x 4 x i32> @llvm.vp.load.nxv4i32.p0(ptr align 4 [[TMP4]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP3]]), !tbaa [[TBAA0:![0-9]+]]
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds [4 x i8], ptr [[B]], i64 [[INDEX]]
+; CHECK-NEXT: [[VP_OP_LOAD4:%.*]] = call <vscale x 4 x i32> @llvm.vp.load.nxv4i32.p0(ptr align 4 [[TMP5]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP3]]), !tbaa [[TBAA0]]
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds [4 x i8], ptr [[C]], i64 [[INDEX]]
+; CHECK-NEXT: [[VP_OP_LOAD5:%.*]] = call <vscale x 4 x i32> @llvm.vp.load.nxv4i32.p0(ptr align 4 [[TMP6]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP3]]), !tbaa [[TBAA0]]
+; CHECK-NEXT: [[TMP7:%.*]] = mul nsw <vscale x 4 x i32> [[VP_OP_LOAD5]], [[VP_OP_LOAD4]]
+; CHECK-NEXT: [[TMP8:%.*]] = add nsw <vscale x 4 x i32> [[TMP7]], [[VP_OP_LOAD]]
+; CHECK-NEXT: call void @llvm.vp.store.nxv4i32.p0(<vscale x 4 x i32> [[TMP8]], ptr align 4 [[TMP4]], <vscale x 4 x i1> splat (i1 true), i32 [[TMP3]]), !tbaa [[TBAA0]]
+; CHECK-NEXT: [[TMP9:%.*]] = zext i32 [[TMP3]] to i64
+; CHECK-NEXT: [[CURRENT_ITERATION_NEXT]] = add i64 [[TMP9]], [[INDEX]]
+; CHECK-NEXT: [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP9]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
+; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br [[L20_LOOPEXIT:label %.*]]
+; CHECK: [[SCALAR_PH]]:
+;
+entry:
+ %cmp.not14 = icmp sgt i32 %n, 0
+ br i1 %cmp.not14, label %if.end.preheader, label %L20
+
+if.end.preheader: ; preds = %entry
+ %0 = sext i32 %inc to i64
+ %1 = zext nneg i32 %n to i64
+ br label %if.end
+
+if.end: ; preds = %if.end.preheader, %if.end
+ %indvars.iv = phi i64 [ 0, %if.end.preheader ], [ %indvars.iv.next, %if.end ]
+ %arrayidx = getelementptr inbounds [4 x i8], ptr %a, i64 %indvars.iv
+ %2 = load i32, ptr %arrayidx, align 4, !tbaa !13
+ %arrayidx2 = getelementptr inbounds [4 x i8], ptr %b, i64 %indvars.iv
+ %3 = load i32, ptr %arrayidx2, align 4, !tbaa !13
+ %arrayidx4 = getelementptr inbounds [4 x i8], ptr %c, i64 %indvars.iv
+ %4 = load i32, ptr %arrayidx4, align 4, !tbaa !13
+ %mul = mul nsw i32 %4, %3
+ %add = add nsw i32 %mul, %2
+ store i32 %add, ptr %arrayidx, align 4, !tbaa !13
+ %indvars.iv.next = add nsw i64 %indvars.iv, %0
+ %cmp.not = icmp slt i64 %indvars.iv.next, %1
+ br i1 %cmp.not, label %if.end, label %L20
+
+L20: ; preds = %if.end, %entry
+ ret void
+}
+
+attributes #0 = { nofree norecurse nosync nounwind memory(argmem: readwrite) vscale_range(4,1024) "target-cpu"="generic-rv64" "target-features"="+64bit,+d,+f,+m,+v,+zicsr,+zve32f,+zve32x,+zve64d,+zve64f,+zve64x,+zvl128b,+zvl32b,+zvl64b" }
+
+!10 = !{!"int", !11, i64 0}
+!11 = !{!"omnipotent char", !12, i64 0}
+!12 = !{!"Simple C/C++ TBAA"}
+!13 = !{!10, !10, i64 0}
``````````
</details>
https://github.com/llvm/llvm-project/pull/222261
More information about the llvm-commits
mailing list