[llvm] [ValueTracking] Don't limit the recursion depth for single-incoming phis in isKnownNonZero (PR #226173)

Konstantin Bogdanov via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 03:40:43 PDT 2026


https://github.com/thevar1able updated https://github.com/llvm/llvm-project/pull/226173

>From cd34def59056c8d7e29d5c23c7782dca7c4919a1 Mon Sep 17 00:00:00 2001
From: Konstantin Bogdanov <konstantin at clickhouse.com>
Date: Thu, 24 Sep 2026 15:44:43 +0200
Subject: [PATCH 1/5] [PhaseOrdering] Precommit test for ctlz loop after an or
 reduction (NFC)

---
 .../X86/ctlz-loop-or-reduction.ll             | 125 ++++++++++++++++++
 1 file changed, 125 insertions(+)
 create mode 100644 llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll

diff --git a/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll b/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll
new file mode 100644
index 0000000000000..1ceae60b4e71c
--- /dev/null
+++ b/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll
@@ -0,0 +1,125 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -O3 -S < %s | FileCheck %s
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+; An or-reduction with a non-zero start value feeds a count-leading-zeros loop
+; behind a zero check. Known bits prove the reduction result is never zero, so
+; InstCombine removes the zero check. LoopIdiomRecognize must still form
+; llvm.ctlz for the unguarded loop (reduced from ffmpeg's
+; sbc_calc_scalefactors).
+;
+; int scalefactor(const int *data, int n) {
+;   unsigned x = 1u << 15;
+;   int i = 0;
+;   do {
+;     x |= abs(data[i]);
+;   } while (++i < n);
+;   int cnt = 32;
+;   while (x) {
+;     x >>= 1;
+;     cnt--;
+;   }
+;   return cnt;
+; }
+
+define i32 @scalefactor(ptr %data, i32 %n) {
+;
+;
+; CHECK-LABEL: define range(i32 -2147483648, 2147483647) i32 @scalefactor(
+; CHECK-SAME: ptr nofree readonly captures(none) [[DATA:%.*]], i32 [[N:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[SMAX:%.*]] = tail call i32 @llvm.smax.i32(i32 [[N]], i32 1)
+; CHECK-NEXT:    [[WIDE_TRIP_COUNT:%.*]] = zext nneg i32 [[SMAX]] to i64
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp slt i32 [[N]], 8
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[RED_LOOP_PREHEADER:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[N_VEC:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 2147483640
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ <i32 32768, i32 0, i32 0, i32 0>, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI2:%.*]] = phi <4 x i32> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DATA]], i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds nuw i8, ptr [[TMP0]], i64 16
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
+; CHECK-NEXT:    [[WIDE_LOAD3:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = tail call <4 x i32> @llvm.abs.v4i32(<4 x i32> [[WIDE_LOAD]], i1 true)
+; CHECK-NEXT:    [[TMP3:%.*]] = tail call <4 x i32> @llvm.abs.v4i32(<4 x i32> [[WIDE_LOAD3]], i1 true)
+; CHECK-NEXT:    [[TMP4]] = or <4 x i32> [[TMP2]], [[VEC_PHI]]
+; CHECK-NEXT:    [[TMP5]] = or <4 x i32> [[TMP3]], [[VEC_PHI2]]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[BIN_RDX:%.*]] = or <4 x i32> [[TMP5]], [[TMP4]]
+; CHECK-NEXT:    [[TMP7:%.*]] = tail call i32 @llvm.vector.reduce.or.v4i32(<4 x i32> [[BIN_RDX]])
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[N_VEC]], [[WIDE_TRIP_COUNT]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[CLZ_LOOP_PREHEADER:.*]], label %[[RED_LOOP_PREHEADER]]
+; CHECK:       [[RED_LOOP_PREHEADER]]:
+; CHECK-NEXT:    [[INDVARS_IV_PH:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[N_VEC]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    [[X_PH:%.*]] = phi i32 [ 32768, %[[ENTRY]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    br label %[[RED_LOOP:.*]]
+; CHECK:       [[RED_LOOP]]:
+; CHECK-NEXT:    [[INDVARS_IV:%.*]] = phi i64 [ [[INDVARS_IV_NEXT:%.*]], %[[RED_LOOP]] ], [ [[INDVARS_IV_PH]], %[[RED_LOOP_PREHEADER]] ]
+; CHECK-NEXT:    [[X:%.*]] = phi i32 [ [[X_NEXT:%.*]], %[[RED_LOOP]] ], [ [[X_PH]], %[[RED_LOOP_PREHEADER]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DATA]], i64 [[INDVARS_IV]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[ABS:%.*]] = tail call i32 @llvm.abs.i32(i32 [[V]], i1 true)
+; CHECK-NEXT:    [[X_NEXT]] = or i32 [[ABS]], [[X]]
+; CHECK-NEXT:    [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
+; CHECK-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], [[WIDE_TRIP_COUNT]]
+; CHECK-NEXT:    br i1 [[EXITCOND_NOT]], label %[[CLZ_LOOP_PREHEADER]], label %[[RED_LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[CLZ_LOOP_PREHEADER]]:
+; CHECK-NEXT:    [[Y_PH:%.*]] = phi i32 [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[X_NEXT]], %[[RED_LOOP]] ]
+; CHECK-NEXT:    br label %[[CLZ_LOOP:.*]]
+; CHECK:       [[CLZ_LOOP]]:
+; CHECK-NEXT:    [[CNT:%.*]] = phi i32 [ [[CNT_NEXT:%.*]], %[[CLZ_LOOP]] ], [ 32, %[[CLZ_LOOP_PREHEADER]] ]
+; CHECK-NEXT:    [[Y:%.*]] = phi i32 [ [[Y_NEXT:%.*]], %[[CLZ_LOOP]] ], [ [[Y_PH]], %[[CLZ_LOOP_PREHEADER]] ]
+; CHECK-NEXT:    [[Y_NEXT]] = lshr i32 [[Y]], 1
+; CHECK-NEXT:    [[CNT_NEXT]] = add nsw i32 [[CNT]], -1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i32 [[Y_NEXT]], 0
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[CLZ_LOOP]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret i32 [[CNT_NEXT]]
+;
+entry:
+  br label %red.loop
+
+red.loop:
+  %i = phi i32 [ 0, %entry ], [ %i.next, %red.loop ]
+  %x = phi i32 [ 32768, %entry ], [ %x.next, %red.loop ]
+  %idx = zext i32 %i to i64
+  %gep = getelementptr inbounds i32, ptr %data, i64 %idx
+  %v = load i32, ptr %gep, align 4
+  %abs = call i32 @llvm.abs.i32(i32 %v, i1 true)
+  %x.next = or i32 %x, %abs
+  %i.next = add nuw nsw i32 %i, 1
+  %ec = icmp slt i32 %i.next, %n
+  br i1 %ec, label %red.loop, label %red.exit
+
+red.exit:
+  %zero = icmp eq i32 %x.next, 0
+  br i1 %zero, label %exit, label %clz.loop
+
+clz.loop:
+  %cnt = phi i32 [ 32, %red.exit ], [ %cnt.next, %clz.loop ]
+  %y = phi i32 [ %x.next, %red.exit ], [ %y.next, %clz.loop ]
+  %y.next = lshr i32 %y, 1
+  %cnt.next = add nsw i32 %cnt, -1
+  %done = icmp eq i32 %y.next, 0
+  br i1 %done, label %exit, label %clz.loop
+
+exit:
+  %cnt.lcssa = phi i32 [ 32, %red.exit ], [ %cnt.next, %clz.loop ]
+  ret i32 %cnt.lcssa
+}
+
+declare i32 @llvm.abs.i32(i32, i1)
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.

>From f212d9afaf4ae0c0e7d87dbbcb8183570c813142 Mon Sep 17 00:00:00 2001
From: Konstantin Bogdanov <konstantin at clickhouse.com>
Date: Thu, 24 Sep 2026 15:49:19 +0200
Subject: [PATCH 2/5] [ValueTracking] Don't limit the recursion depth for
 single-incoming phis in isKnownNonZero

isKnownNonZero only lets a phi recurse into its incoming values at the
last level of depth. A phi with a single incoming value is just a copy,
and LCSSA puts such a phi between a loop result and its users, so this
hides facts that are otherwise available: after InstCombine removes the
zero check in front of a clz loop, LoopIdiomRecognize can no longer
prove that the value reaching the loop through the LCSSA phi is non-zero
and doesn't form llvm.ctlz. Keep the current depth for such phis.

Reduced from ffmpeg's sbc_calc_scalefactors.
---
 llvm/lib/Analysis/ValueTracking.cpp                |  7 +++++--
 .../PhaseOrdering/X86/ctlz-loop-or-reduction.ll    | 14 +++-----------
 2 files changed, 8 insertions(+), 13 deletions(-)

diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index dd5f6fc7ad16a..f4deea7694ec0 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -3557,9 +3557,12 @@ static bool isKnownNonZeroFromOperator(const Operator *I,
     if (Q.IIQ.UseInstrInfo && isNonZeroRecurrence(PN))
       return true;
 
-    // Check if all incoming values are non-zero using recursion.
+    // Check if all incoming values are non-zero using recursion. A phi with a
+    // single incoming value is just a copy, so don't limit the depth for it.
     SimplifyQuery RecQ = Q.getWithoutCondContext();
-    unsigned NewDepth = std::max(Depth, MaxAnalysisRecursionDepth - 1);
+    unsigned NewDepth = PN->getNumIncomingValues() == 1
+                            ? Depth
+                            : std::max(Depth, MaxAnalysisRecursionDepth - 1);
     return llvm::all_of(PN->operands(), [&](const Use &U) {
       if (U.get() == PN)
         return true;
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll b/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll
index 1ceae60b4e71c..a727e941bcca0 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/ctlz-loop-or-reduction.ll
@@ -27,7 +27,7 @@ target triple = "x86_64-unknown-linux-gnu"
 define i32 @scalefactor(ptr %data, i32 %n) {
 ;
 ;
-; CHECK-LABEL: define range(i32 -2147483648, 2147483647) i32 @scalefactor(
+; CHECK-LABEL: define range(i32 1, 17) i32 @scalefactor(
 ; CHECK-SAME: ptr nofree readonly captures(none) [[DATA:%.*]], i32 [[N:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
 ; CHECK-NEXT:    [[SMAX:%.*]] = tail call i32 @llvm.smax.i32(i32 [[N]], i32 1)
@@ -72,16 +72,8 @@ define i32 @scalefactor(ptr %data, i32 %n) {
 ; CHECK-NEXT:    [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], [[WIDE_TRIP_COUNT]]
 ; CHECK-NEXT:    br i1 [[EXITCOND_NOT]], label %[[CLZ_LOOP_PREHEADER]], label %[[RED_LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
 ; CHECK:       [[CLZ_LOOP_PREHEADER]]:
-; CHECK-NEXT:    [[Y_PH:%.*]] = phi i32 [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[X_NEXT]], %[[RED_LOOP]] ]
-; CHECK-NEXT:    br label %[[CLZ_LOOP:.*]]
-; CHECK:       [[CLZ_LOOP]]:
-; CHECK-NEXT:    [[CNT:%.*]] = phi i32 [ [[CNT_NEXT:%.*]], %[[CLZ_LOOP]] ], [ 32, %[[CLZ_LOOP_PREHEADER]] ]
-; CHECK-NEXT:    [[Y:%.*]] = phi i32 [ [[Y_NEXT:%.*]], %[[CLZ_LOOP]] ], [ [[Y_PH]], %[[CLZ_LOOP_PREHEADER]] ]
-; CHECK-NEXT:    [[Y_NEXT]] = lshr i32 [[Y]], 1
-; CHECK-NEXT:    [[CNT_NEXT]] = add nsw i32 [[CNT]], -1
-; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i32 [[Y_NEXT]], 0
-; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[CLZ_LOOP]]
-; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[X_NEXT_LCSSA:%.*]] = phi i32 [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ [[X_NEXT]], %[[RED_LOOP]] ]
+; CHECK-NEXT:    [[CNT_NEXT:%.*]] = tail call range(i32 0, 33) i32 @llvm.ctlz.i32(i32 [[X_NEXT_LCSSA]], i1 true)
 ; CHECK-NEXT:    ret i32 [[CNT_NEXT]]
 ;
 entry:

>From e7c86924fc8f8f49bdc2902815dc3a3d86c11711 Mon Sep 17 00:00:00 2001
From: Konstantin Bogdanov <konstantin at clickhouse.com>
Date: Thu, 24 Sep 2026 16:20:00 +0200
Subject: [PATCH 3/5] [LoopIdiom] Precommit test for ctlz of a value reaching
 the loop through an LCSSA phi (NFC)

---
 .../LoopIdiom/X86/ctlz-lcssa-phi.ll           | 70 +++++++++++++++++++
 1 file changed, 70 insertions(+)
 create mode 100644 llvm/test/Transforms/LoopIdiom/X86/ctlz-lcssa-phi.ll

diff --git a/llvm/test/Transforms/LoopIdiom/X86/ctlz-lcssa-phi.ll b/llvm/test/Transforms/LoopIdiom/X86/ctlz-lcssa-phi.ll
new file mode 100644
index 0000000000000..5ab87c50b93e9
--- /dev/null
+++ b/llvm/test/Transforms/LoopIdiom/X86/ctlz-lcssa-phi.ll
@@ -0,0 +1,70 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-idiom -S < %s | FileCheck %s
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+; The input of the clz loop comes through an LCSSA phi of an or recurrence
+; with a non-zero start value, so it is non-zero and the loop can be
+; converted to llvm.ctlz without a zero check.
+define i32 @ctlz_lcssa_or_recurrence(ptr %p, i32 %n) {
+; CHECK-LABEL: define i32 @ctlz_lcssa_or_recurrence(
+; CHECK-SAME: ptr [[P:%.*]], i32 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[RED_LOOP:.*]]
+; CHECK:       [[RED_LOOP]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[RED_LOOP]] ]
+; CHECK-NEXT:    [[X:%.*]] = phi i32 [ 32768, %[[ENTRY]] ], [ [[X_NEXT:%.*]], %[[RED_LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[P]], i32 [[I]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[ABS:%.*]] = call i32 @llvm.abs.i32(i32 [[V]], i1 true)
+; CHECK-NEXT:    [[X_NEXT]] = or i32 [[ABS]], [[X]]
+; CHECK-NEXT:    [[I_NEXT]] = add nuw nsw i32 [[I]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp slt i32 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[RED_LOOP]], label %[[CLZ_PH:.*]]
+; CHECK:       [[CLZ_PH]]:
+; CHECK-NEXT:    [[X_LCSSA:%.*]] = phi i32 [ [[X_NEXT]], %[[RED_LOOP]] ]
+; CHECK-NEXT:    br label %[[CLZ_LOOP:.*]]
+; CHECK:       [[CLZ_LOOP]]:
+; CHECK-NEXT:    [[CNT:%.*]] = phi i32 [ 32, %[[CLZ_PH]] ], [ [[CNT_NEXT:%.*]], %[[CLZ_LOOP]] ]
+; CHECK-NEXT:    [[Y:%.*]] = phi i32 [ [[X_LCSSA]], %[[CLZ_PH]] ], [ [[Y_NEXT:%.*]], %[[CLZ_LOOP]] ]
+; CHECK-NEXT:    [[Y_NEXT]] = lshr i32 [[Y]], 1
+; CHECK-NEXT:    [[CNT_NEXT]] = add nsw i32 [[CNT]], -1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i32 [[Y_NEXT]], 0
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[CLZ_LOOP]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[CNT_LCSSA:%.*]] = phi i32 [ [[CNT_NEXT]], %[[CLZ_LOOP]] ]
+; CHECK-NEXT:    ret i32 [[CNT_LCSSA]]
+;
+entry:
+  br label %red.loop
+
+red.loop:
+  %i = phi i32 [ 0, %entry ], [ %i.next, %red.loop ]
+  %x = phi i32 [ 32768, %entry ], [ %x.next, %red.loop ]
+  %gep = getelementptr inbounds i32, ptr %p, i32 %i
+  %v = load i32, ptr %gep, align 4
+  %abs = call i32 @llvm.abs.i32(i32 %v, i1 true)
+  %x.next = or i32 %abs, %x
+  %i.next = add nuw nsw i32 %i, 1
+  %ec = icmp slt i32 %i.next, %n
+  br i1 %ec, label %red.loop, label %clz.ph
+
+clz.ph:
+  %x.lcssa = phi i32 [ %x.next, %red.loop ]
+  br label %clz.loop
+
+clz.loop:
+  %cnt = phi i32 [ 32, %clz.ph ], [ %cnt.next, %clz.loop ]
+  %y = phi i32 [ %x.lcssa, %clz.ph ], [ %y.next, %clz.loop ]
+  %y.next = lshr i32 %y, 1
+  %cnt.next = add nsw i32 %cnt, -1
+  %done = icmp eq i32 %y.next, 0
+  br i1 %done, label %exit, label %clz.loop
+
+exit:
+  %cnt.lcssa = phi i32 [ %cnt.next, %clz.loop ]
+  ret i32 %cnt.lcssa
+}
+
+declare i32 @llvm.abs.i32(i32, i1)

>From 0b359bdc006c714275bd2ea4ab59b7db098a1a43 Mon Sep 17 00:00:00 2001
From: Konstantin Bogdanov <konstantin at clickhouse.com>
Date: Thu, 24 Sep 2026 16:28:12 +0200
Subject: [PATCH 4/5] [ValueTracking] Don't limit the recursion depth for
 single-incoming phis everywhere

Apply the same rule as in isKnownNonZero to the other places that cap
the recursion depth for phi operands: computeKnownBits,
isKnownToBeAPowerOfTwo and computeKnownFPClass. A phi with a single
incoming value is just a copy, as created by LCSSA, so recurse into it
with the usual depth increment instead.

Add unit tests for all four analyses, since InstSimplify and InstCombine
fold such phis before querying them.
---
 llvm/lib/Analysis/ValueTracking.cpp           | 24 ++++---
 .../LoopIdiom/X86/ctlz-lcssa-phi.ll           |  9 ++-
 .../Transforms/PhaseOrdering/X86/pr38280.ll   |  2 +-
 llvm/unittests/Analysis/ValueTrackingTest.cpp | 67 +++++++++++++++++++
 4 files changed, 91 insertions(+), 11 deletions(-)

diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index f4deea7694ec0..1e4b1aea1a53c 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -1998,8 +1998,11 @@ static void computeKnownBitsFromOperator(const Operator *I,
       break;
 
     // Otherwise take the unions of the known bit sets of the operands,
-    // taking conservative care to avoid excessive recursion.
-    if (Depth < MaxAnalysisRecursionDepth - 1 && Known.isUnknown()) {
+    // taking conservative care to avoid excessive recursion. A phi with a
+    // single incoming value is just a copy, so don't limit the depth for it.
+    bool IsCopy = P->getNumIncomingValues() == 1;
+    if ((IsCopy || Depth < MaxAnalysisRecursionDepth - 1) &&
+        Known.isUnknown()) {
       // Skip if every incoming value references to ourself.
       if (isa_and_nonnull<UndefValue>(P->hasConstantValue()))
         break;
@@ -2027,7 +2030,7 @@ static void computeKnownBitsFromOperator(const Operator *I,
         // TODO: See if we can base recursion limiter on number of incoming phi
         // edges so we don't overly clamp analysis.
         computeKnownBits(IncValue, DemandedElts, Known2, RecQ,
-                         MaxAnalysisRecursionDepth - 1);
+                         IsCopy ? Depth + 1 : MaxAnalysisRecursionDepth - 1);
 
         // See if we can further use a conditional branch into the phi
         // to help us determine the range of the value.
@@ -2917,8 +2920,11 @@ bool llvm::isKnownToBeAPowerOfTwo(const Value *V, bool OrZero,
       return true;
 
     // Recursively check all incoming values. Limit recursion to 2 levels, so
-    // that search complexity is limited to number of operands^2.
-    unsigned NewDepth = std::max(Depth, MaxAnalysisRecursionDepth - 1);
+    // that search complexity is limited to number of operands^2. A phi with a
+    // single incoming value is just a copy, so don't limit the depth for it.
+    unsigned NewDepth = PN->getNumIncomingValues() == 1
+                            ? Depth
+                            : std::max(Depth, MaxAnalysisRecursionDepth - 1);
     return llvm::all_of(PN->operands(), [&](const Use &U) {
       // Value is power of 2 if it is coming from PHI node itself by induction.
       if (U.get() == PN)
@@ -6370,10 +6376,12 @@ void computeKnownFPClass(const Value *V, const APInt &DemandedElts,
       break;
 
     // Otherwise take the unions of the known bit sets of the operands,
-    // taking conservative care to avoid excessive recursion.
+    // taking conservative care to avoid excessive recursion. A phi with a
+    // single incoming value is just a copy, so don't limit the depth for it.
     const unsigned PhiRecursionLimit = MaxAnalysisRecursionDepth - 2;
+    bool IsCopy = P->getNumIncomingValues() == 1;
 
-    if (Depth < PhiRecursionLimit) {
+    if (IsCopy || Depth < PhiRecursionLimit) {
       // Skip if every incoming value references to ourself.
       if (isa_and_nonnull<UndefValue>(P->hasConstantValue()))
         break;
@@ -6394,7 +6402,7 @@ void computeKnownFPClass(const Value *V, const APInt &DemandedElts,
         // detect known sign bits.
         computeKnownFPClass(IncValue, DemandedElts, InterestedClasses, KnownSrc,
                             Q.getWithoutCondContext().getWithInstruction(CtxI),
-                            PhiRecursionLimit);
+                            IsCopy ? Depth + 1 : PhiRecursionLimit);
 
         if (First) {
           Known = KnownSrc;
diff --git a/llvm/test/Transforms/LoopIdiom/X86/ctlz-lcssa-phi.ll b/llvm/test/Transforms/LoopIdiom/X86/ctlz-lcssa-phi.ll
index 5ab87c50b93e9..5f12583f42ce0 100644
--- a/llvm/test/Transforms/LoopIdiom/X86/ctlz-lcssa-phi.ll
+++ b/llvm/test/Transforms/LoopIdiom/X86/ctlz-lcssa-phi.ll
@@ -24,16 +24,21 @@ define i32 @ctlz_lcssa_or_recurrence(ptr %p, i32 %n) {
 ; CHECK-NEXT:    br i1 [[EC]], label %[[RED_LOOP]], label %[[CLZ_PH:.*]]
 ; CHECK:       [[CLZ_PH]]:
 ; CHECK-NEXT:    [[X_LCSSA:%.*]] = phi i32 [ [[X_NEXT]], %[[RED_LOOP]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = call i32 @llvm.ctlz.i32(i32 [[X_LCSSA]], i1 true)
+; CHECK-NEXT:    [[TMP1:%.*]] = sub i32 32, [[TMP0]]
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 32, [[TMP1]]
 ; CHECK-NEXT:    br label %[[CLZ_LOOP:.*]]
 ; CHECK:       [[CLZ_LOOP]]:
+; CHECK-NEXT:    [[TCPHI:%.*]] = phi i32 [ [[TMP1]], %[[CLZ_PH]] ], [ [[TCDEC:%.*]], %[[CLZ_LOOP]] ]
 ; CHECK-NEXT:    [[CNT:%.*]] = phi i32 [ 32, %[[CLZ_PH]] ], [ [[CNT_NEXT:%.*]], %[[CLZ_LOOP]] ]
 ; CHECK-NEXT:    [[Y:%.*]] = phi i32 [ [[X_LCSSA]], %[[CLZ_PH]] ], [ [[Y_NEXT:%.*]], %[[CLZ_LOOP]] ]
 ; CHECK-NEXT:    [[Y_NEXT]] = lshr i32 [[Y]], 1
 ; CHECK-NEXT:    [[CNT_NEXT]] = add nsw i32 [[CNT]], -1
-; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i32 [[Y_NEXT]], 0
+; CHECK-NEXT:    [[TCDEC]] = sub nsw i32 [[TCPHI]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i32 [[TCDEC]], 0
 ; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[CLZ_LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[CNT_LCSSA:%.*]] = phi i32 [ [[CNT_NEXT]], %[[CLZ_LOOP]] ]
+; CHECK-NEXT:    [[CNT_LCSSA:%.*]] = phi i32 [ [[TMP2]], %[[CLZ_LOOP]] ]
 ; CHECK-NEXT:    ret i32 [[CNT_LCSSA]]
 ;
 entry:
diff --git a/llvm/test/Transforms/PhaseOrdering/X86/pr38280.ll b/llvm/test/Transforms/PhaseOrdering/X86/pr38280.ll
index 43f6cf7e51d09..a0c79a30ac9d7 100644
--- a/llvm/test/Transforms/PhaseOrdering/X86/pr38280.ll
+++ b/llvm/test/Transforms/PhaseOrdering/X86/pr38280.ll
@@ -32,7 +32,7 @@ define void @apply_delta(ptr nocapture noundef %dst, ptr nocapture noundef reado
 ; CHECK-NEXT:    [[DST_ADDR_130:%.*]] = phi ptr [ [[INCDEC_PTR:%.*]], [[WHILE_BODY4]] ], [ [[DST_ADDR_0_LCSSA]], [[WHILE_COND3_PREHEADER]] ]
 ; CHECK-NEXT:    [[SRC_ADDR_129:%.*]] = phi ptr [ [[INCDEC_PTR8:%.*]], [[WHILE_BODY4]] ], [ [[SRC_ADDR_0_LCSSA]], [[WHILE_COND3_PREHEADER]] ]
 ; CHECK-NEXT:    [[COUNT_ADDR_128:%.*]] = phi i64 [ [[DEC:%.*]], [[WHILE_BODY4]] ], [ [[COUNT_ADDR_0_LCSSA]], [[WHILE_COND3_PREHEADER]] ]
-; CHECK-NEXT:    [[DEC]] = add i64 [[COUNT_ADDR_128]], -1
+; CHECK-NEXT:    [[DEC]] = add nsw i64 [[COUNT_ADDR_128]], -1
 ; CHECK-NEXT:    [[TMP2:%.*]] = load i8, ptr [[SRC_ADDR_129]], align 1
 ; CHECK-NEXT:    [[ARRAYIDX:%.*]] = getelementptr inbounds i8, ptr [[DST_ADDR_130]], i64 [[NEG_OFFS]]
 ; CHECK-NEXT:    [[TMP3:%.*]] = load i8, ptr [[ARRAYIDX]], align 1
diff --git a/llvm/unittests/Analysis/ValueTrackingTest.cpp b/llvm/unittests/Analysis/ValueTrackingTest.cpp
index 6458f71453abe..0143ef81f88b0 100644
--- a/llvm/unittests/Analysis/ValueTrackingTest.cpp
+++ b/llvm/unittests/Analysis/ValueTrackingTest.cpp
@@ -2132,6 +2132,25 @@ TEST_F(ComputeKnownFPClassTest, SelfPhiSecondArg) {
   expectKnownFPClass(~fcInf, std::nullopt);
 }
 
+// A phi with a single incoming value, as created by LCSSA, doesn't limit the
+// recursion depth.
+TEST_F(ComputeKnownFPClassTest, PhiSingleIncoming) {
+  parseAssembly("declare float @llvm.fabs.f32(float)\n"
+                "define float @test(float %arg, i1 %c) {\n"
+                "entry:\n"
+                "  br label %loop\n"
+                "loop:\n"
+                "  %a = call float @llvm.fabs.f32(float %arg)\n"
+                "  %b = fneg float %a\n"
+                "  %d = fneg float %b\n"
+                "  br i1 %c, label %loop, label %exit\n"
+                "exit:\n"
+                "  %A = phi float [ %d, %loop ]\n"
+                "  ret float %A\n"
+                "}\n");
+  expectKnownFPClass(fcPositive | fcNan, false);
+}
+
 TEST_F(ComputeKnownFPClassTest, CannotBeOrderedLessThanZero) {
   parseAssembly("define float @test(float %arg) {\n"
                 "  %A = fmul float %arg, %arg"
@@ -2734,6 +2753,54 @@ TEST_F(ValueTrackingTest, IsImpliedConditionBitMaskSigned) {
   EXPECT_EQ(isImpliedCondition(A, A2, DL, true), true);
 }
 
+// A phi with a single incoming value, as created by LCSSA, doesn't limit the
+// recursion depth.
+TEST_F(ComputeKnownBitsTest, ComputeKnownBitsPhiSingleIncoming) {
+  parseAssembly("define i32 @test(i32 %x, i1 %c) {\n"
+                "entry:\n"
+                "  br label %loop\n"
+                "loop:\n"
+                "  %a = or i32 %x, 1\n"
+                "  %b = shl i32 %a, 2\n"
+                "  br i1 %c, label %loop, label %exit\n"
+                "exit:\n"
+                "  %A = phi i32 [ %b, %loop ]\n"
+                "  ret i32 %A\n"
+                "}\n");
+  expectKnownBits(/*Zero*/ 3u, /*One*/ 4u);
+}
+
+TEST_F(ComputeKnownBitsTest, KnownNonZeroPhiSingleIncoming) {
+  parseAssembly("define i32 @test(i32 %x, i1 %c) {\n"
+                "entry:\n"
+                "  br label %loop\n"
+                "loop:\n"
+                "  %a = or i32 %x, 1\n"
+                "  %b = shl nuw i32 %a, 2\n"
+                "  br i1 %c, label %loop, label %exit\n"
+                "exit:\n"
+                "  %A = phi i32 [ %b, %loop ]\n"
+                "  ret i32 %A\n"
+                "}\n");
+  EXPECT_TRUE(isKnownNonZero(A, SimplifyQuery(M->getDataLayout())));
+}
+
+TEST_F(ComputeKnownBitsTest, KnownPowerOfTwoPhiSingleIncoming) {
+  parseAssembly("define i32 @test(i32 %x, i1 %c) {\n"
+                "entry:\n"
+                "  br label %loop\n"
+                "loop:\n"
+                "  %a = shl i32 1, %x\n"
+                "  %b = shl nuw i32 %a, 1\n"
+                "  %d = shl nuw i32 %b, 1\n"
+                "  br i1 %c, label %loop, label %exit\n"
+                "exit:\n"
+                "  %A = phi i32 [ %d, %loop ]\n"
+                "  ret i32 %A\n"
+                "}\n");
+  EXPECT_TRUE(isKnownToBeAPowerOfTwo(A, M->getDataLayout()));
+}
+
 TEST_F(ComputeKnownBitsTest, KnownNonZeroShift) {
   // %q is known nonzero without known bits.
   // Because %q is nonzero, %A[0] is known to be zero.

>From 83172b0aadd1f71a5e6684c6caf1125d7f28aa32 Mon Sep 17 00:00:00 2001
From: Konstantin Bogdanov <konstantin at clickhouse.com>
Date: Thu, 24 Sep 2026 18:34:23 +0200
Subject: [PATCH 5/5] Move the LoopIdiom test out of X86, it doesn't need a
 target

---
 .../Transforms/LoopIdiom/{X86 => }/ctlz-lcssa-phi.ll     | 9 +++------
 1 file changed, 3 insertions(+), 6 deletions(-)
 rename llvm/test/Transforms/LoopIdiom/{X86 => }/ctlz-lcssa-phi.ll (89%)

diff --git a/llvm/test/Transforms/LoopIdiom/X86/ctlz-lcssa-phi.ll b/llvm/test/Transforms/LoopIdiom/ctlz-lcssa-phi.ll
similarity index 89%
rename from llvm/test/Transforms/LoopIdiom/X86/ctlz-lcssa-phi.ll
rename to llvm/test/Transforms/LoopIdiom/ctlz-lcssa-phi.ll
index 5f12583f42ce0..5581258ecb107 100644
--- a/llvm/test/Transforms/LoopIdiom/X86/ctlz-lcssa-phi.ll
+++ b/llvm/test/Transforms/LoopIdiom/ctlz-lcssa-phi.ll
@@ -1,12 +1,9 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt -passes=loop-idiom -S < %s | FileCheck %s
 
-target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
-target triple = "x86_64-unknown-linux-gnu"
-
-; The input of the clz loop comes through an LCSSA phi of an or recurrence
-; with a non-zero start value, so it is non-zero and the loop can be
-; converted to llvm.ctlz without a zero check.
+; The input of the clz loop comes through the single-incoming LCSSA phi
+; %x.lcssa of an or recurrence with a non-zero start value, so it is non-zero
+; and the loop can be converted to llvm.ctlz without a zero check.
 define i32 @ctlz_lcssa_or_recurrence(ptr %p, i32 %n) {
 ; CHECK-LABEL: define i32 @ctlz_lcssa_or_recurrence(
 ; CHECK-SAME: ptr [[P:%.*]], i32 [[N:%.*]]) {



More information about the llvm-commits mailing list