[llvm] [ValueTracking] Don't limit the recursion depth for single-incoming phis in isKnownNonZero (PR #226173)
Konstantin Bogdanov via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 07:04:19 PDT 2026
https://github.com/thevar1able created https://github.com/llvm/llvm-project/pull/226173
`isKnownNonZero` only lets a phi recurse into its incoming values at the last level of depth. A phi with a single incoming value is just a copy, and LCSSA puts one between a loop result and its users. After InstCombine removes the zero check in front of a clz loop, LoopIdiomRecognize can then no longer prove that the value reaching the loop through the LCSSA phi is non-zero, and it doesn't form `llvm.ctlz`. This keeps the current depth for such phis.
`PhaseOrdering/X86/ctlz-loop-or-reduction.ll` (reduced from ffmpeg's `sbc_calc_scalefactors`) covers the end-to-end result, and the first commit precommits it.
Split out of #222334. Depends on #226155.
>From 9341b71fa7e61f19a628ca204228f9dc6816b595 Mon Sep 17 00:00:00 2001
From: Konstantin Bogdanov <konstantin at clickhouse.com>
Date: Thu, 24 Sep 2026 15:44:43 +0200
Subject: [PATCH 1/2] [PhaseOrdering] Precommit test for ctlz loop after an or
reduction (NFC)
---
.../X86/ctlz-loop-or-reduction.ll | 125 ++++++++++++++++++
1 file changed, 125 insertions(+)
create mode 100644 llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll b/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll
new file mode 100644
index 0000000000000..1ceae60b4e71c
--- /dev/null
+++ b/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll
@@ -0,0 +1,125 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -O3 -S < %s | FileCheck %s
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+; An or-reduction with a non-zero start value feeds a count-leading-zeros loop
+; behind a zero check. Known bits prove the reduction result is never zero, so
+; InstCombine removes the zero check. LoopIdiomRecognize must still form
+; llvm.ctlz for the unguarded loop (reduced from ffmpeg's
+; sbc_calc_scalefactors).
+;
+; int scalefactor(const int *data, int n) {
+; unsigned x = 1u << 15;
+; int i = 0;
+; do {
+; x |= abs(data[i]);
+; } while (++i < n);
+; int cnt = 32;
+; while (x) {
+; x >>= 1;
+; cnt--;
+; }
+; return cnt;
+; }
+
+define i32 @scalefactor(ptr %data, i32 %n) {
+;
+;
+; CHECK-LABEL: define range(i32 -2147483648, 2147483647) i32 @scalefactor(
+; CHECK-SAME: ptr nofree readonly captures(none) [[DATA:%.*]], i32 [[N:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[SMAX:%.*]] = tail call i32 @llvm.smax.i32(i32 [[N]], i32 1)
+; CHECK-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext nneg i32 [[SMAX]] to i64
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp slt i32 [[N]], 8
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[RED_LOOP_PREHEADER:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_VEC:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 2147483640
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i32> [ <i32 32768, i32 0, i32 0, i32 0>, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DATA]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 16
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
+; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = tail call <4 x i32> @llvm.abs.v4i32(<4 x i32> [[WIDE_LOAD]], i1 true)
+; CHECK-NEXT: [[TMP3:%.*]] = tail call <4 x i32> @llvm.abs.v4i32(<4 x i32> [[WIDE_LOAD3]], i1 true)
+; CHECK-NEXT: [[TMP4]] = or <4 x i32> [[TMP2]], [[VEC_PHI]]
+; CHECK-NEXT: [[TMP5]] = or <4 x i32> [[TMP3]], [[VEC_PHI2]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[BIN_RDX:%.*]] = or <4 x i32> [[TMP5]], [[TMP4]]
+; CHECK-NEXT: [[TMP7:%.*]] = tail call i32 @llvm.vector.reduce.or.v4i32(<4 x i32> [[BIN_RDX]])
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[N_VEC]], [[WIDE_TRIP_COUNT]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[CLZ_LOOP_PREHEADER:.*]], label %[[RED_LOOP_PREHEADER]]
+; CHECK: [[RED_LOOP_PREHEADER]]:
+; CHECK-NEXT: [[INDVARS_IV_PH:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[N_VEC]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: [[X_PH:%.*]] = phi i32 [ 32768, %[[ENTRY]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: br label %[[RED_LOOP:.*]]
+; CHECK: [[RED_LOOP]]:
+; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ [[INDVARS_IV_NEXT:%.*]], %[[RED_LOOP]] ], [ [[INDVARS_IV_PH]], %[[RED_LOOP_PREHEADER]] ]
+; CHECK-NEXT: [[X:%.*]] = phi i32 [ [[X_NEXT:%.*]], %[[RED_LOOP]] ], [ [[X_PH]], %[[RED_LOOP_PREHEADER]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DATA]], i64 [[INDVARS_IV]]
+; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT: [[ABS:%.*]] = tail call i32 @llvm.abs.i32(i32 [[V]], i1 true)
+; CHECK-NEXT: [[X_NEXT]] = or i32 [[ABS]], [[X]]
+; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
+; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], [[WIDE_TRIP_COUNT]]
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label %[[CLZ_LOOP_PREHEADER]], label %[[RED_LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[CLZ_LOOP_PREHEADER]]:
+; CHECK-NEXT: [[Y_PH:%.*]] = phi i32 [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[X_NEXT]], %[[RED_LOOP]] ]
+; CHECK-NEXT: br label %[[CLZ_LOOP:.*]]
+; CHECK: [[CLZ_LOOP]]:
+; CHECK-NEXT: [[CNT:%.*]] = phi i32 [ [[CNT_NEXT:%.*]], %[[CLZ_LOOP]] ], [ 32, %[[CLZ_LOOP_PREHEADER]] ]
+; CHECK-NEXT: [[Y:%.*]] = phi i32 [ [[Y_NEXT:%.*]], %[[CLZ_LOOP]] ], [ [[Y_PH]], %[[CLZ_LOOP_PREHEADER]] ]
+; CHECK-NEXT: [[Y_NEXT]] = lshr i32 [[Y]], 1
+; CHECK-NEXT: [[CNT_NEXT]] = add nsw i32 [[CNT]], -1
+; CHECK-NEXT: [[DONE:%.*]] = icmp eq i32 [[Y_NEXT]], 0
+; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT:.*]], label %[[CLZ_LOOP]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret i32 [[CNT_NEXT]]
+;
+entry:
+ br label %red.loop
+
+red.loop:
+ %i = phi i32 [ 0, %entry ], [ %i.next, %red.loop ]
+ %x = phi i32 [ 32768, %entry ], [ %x.next, %red.loop ]
+ %idx = zext i32 %i to i64
+ %gep = getelementptr inbounds i32, ptr %data, i64 %idx
+ %v = load i32, ptr %gep, align 4
+ %abs = call i32 @llvm.abs.i32(i32 %v, i1 true)
+ %x.next = or i32 %x, %abs
+ %i.next = add nuw nsw i32 %i, 1
+ %ec = icmp slt i32 %i.next, %n
+ br i1 %ec, label %red.loop, label %red.exit
+
+red.exit:
+ %zero = icmp eq i32 %x.next, 0
+ br i1 %zero, label %exit, label %clz.loop
+
+clz.loop:
+ %cnt = phi i32 [ 32, %red.exit ], [ %cnt.next, %clz.loop ]
+ %y = phi i32 [ %x.next, %red.exit ], [ %y.next, %clz.loop ]
+ %y.next = lshr i32 %y, 1
+ %cnt.next = add nsw i32 %cnt, -1
+ %done = icmp eq i32 %y.next, 0
+ br i1 %done, label %exit, label %clz.loop
+
+exit:
+ %cnt.lcssa = phi i32 [ 32, %red.exit ], [ %cnt.next, %clz.loop ]
+ ret i32 %cnt.lcssa
+}
+
+declare i32 @llvm.abs.i32(i32, i1)
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.
>From e788f45d4af5403bf47bb8268638ca10f87a7808 Mon Sep 17 00:00:00 2001
From: Konstantin Bogdanov <konstantin at clickhouse.com>
Date: Thu, 24 Sep 2026 15:49:19 +0200
Subject: [PATCH 2/2] [ValueTracking] Don't limit the recursion depth for
single-incoming phis in isKnownNonZero
isKnownNonZero only lets a phi recurse into its incoming values at the
last level of depth. A phi with a single incoming value is just a copy,
and LCSSA puts such a phi between a loop result and its users, so this
hides facts that are otherwise available: after InstCombine removes the
zero check in front of a clz loop, LoopIdiomRecognize can no longer
prove that the value reaching the loop through the LCSSA phi is non-zero
and doesn't form llvm.ctlz. Keep the current depth for such phis.
Reduced from ffmpeg's sbc_calc_scalefactors.
---
llvm/lib/Analysis/ValueTracking.cpp | 7 +++++--
.../PhaseOrdering/X86/ctlz-loop-or-reduction.ll | 14 +++-----------
2 files changed, 8 insertions(+), 13 deletions(-)
diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index 6a54ec8ecc5d6..1bd51d25597b5 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -3557,9 +3557,12 @@ static bool isKnownNonZeroFromOperator(const Operator *I,
if (Q.IIQ.UseInstrInfo && isNonZeroRecurrence(PN))
return true;
- // Check if all incoming values are non-zero using recursion.
+ // Check if all incoming values are non-zero using recursion. A phi with a
+ // single incoming value is just a copy, so don't limit the depth for it.
SimplifyQuery RecQ = Q.getWithoutCondContext();
- unsigned NewDepth = std::max(Depth, MaxAnalysisRecursionDepth - 1);
+ unsigned NewDepth = PN->getNumIncomingValues() == 1
+ ? Depth
+ : std::max(Depth, MaxAnalysisRecursionDepth - 1);
return llvm::all_of(PN->operands(), [&](const Use &U) {
if (U.get() == PN)
return true;
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll b/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll
index 1ceae60b4e71c..a727e941bcca0 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll
@@ -27,7 +27,7 @@ target triple = "x86_64-unknown-linux-gnu"
define i32 @scalefactor(ptr %data, i32 %n) {
;
;
-; CHECK-LABEL: define range(i32 -2147483648, 2147483647) i32 @scalefactor(
+; CHECK-LABEL: define range(i32 1, 17) i32 @scalefactor(
; CHECK-SAME: ptr nofree readonly captures(none) [[DATA:%.*]], i32 [[N:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[SMAX:%.*]] = tail call i32 @llvm.smax.i32(i32 [[N]], i32 1)
@@ -72,16 +72,8 @@ define i32 @scalefactor(ptr %data, i32 %n) {
; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], [[WIDE_TRIP_COUNT]]
; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label %[[CLZ_LOOP_PREHEADER]], label %[[RED_LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: [[CLZ_LOOP_PREHEADER]]:
-; CHECK-NEXT: [[Y_PH:%.*]] = phi i32 [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[X_NEXT]], %[[RED_LOOP]] ]
-; CHECK-NEXT: br label %[[CLZ_LOOP:.*]]
-; CHECK: [[CLZ_LOOP]]:
-; CHECK-NEXT: [[CNT:%.*]] = phi i32 [ [[CNT_NEXT:%.*]], %[[CLZ_LOOP]] ], [ 32, %[[CLZ_LOOP_PREHEADER]] ]
-; CHECK-NEXT: [[Y:%.*]] = phi i32 [ [[Y_NEXT:%.*]], %[[CLZ_LOOP]] ], [ [[Y_PH]], %[[CLZ_LOOP_PREHEADER]] ]
-; CHECK-NEXT: [[Y_NEXT]] = lshr i32 [[Y]], 1
-; CHECK-NEXT: [[CNT_NEXT]] = add nsw i32 [[CNT]], -1
-; CHECK-NEXT: [[DONE:%.*]] = icmp eq i32 [[Y_NEXT]], 0
-; CHECK-NEXT: br i1 [[DONE]], label %[[EXIT:.*]], label %[[CLZ_LOOP]]
-; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[X_NEXT_LCSSA:%.*]] = phi i32 [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[X_NEXT]], %[[RED_LOOP]] ]
+; CHECK-NEXT: [[CNT_NEXT:%.*]] = tail call range(i32 0, 33) i32 @llvm.ctlz.i32(i32 [[X_NEXT_LCSSA]], i1 true)
; CHECK-NEXT: ret i32 [[CNT_NEXT]]
;
entry:
More information about the llvm-commits
mailing list