[llvm] [ValueTracking] Strip zero tests through common zero-test roots (PR #208379)

via llvm-commits llvm-commits at lists.llvm.org
Wed Jul 8 22:22:44 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-analysis

Author: Jaeuk Lee (skku970412)

<details>
<summary>Changes</summary>

Summary:
- Rework `stripNullTest` around a small zero-test-root canonicalization.
- Add a shared `getOperandWithSameZeroTest()` helper for lane-preserving intrinsics whose per-lane zero-test is equivalent to operand 0: `ctpop`, `bswap`, and `bitreverse`.
- Reuse that helper in `isKnownNonZero()` for the same intrinsic knowledge.
- Model the OR zero-test rule only when both sides reduce to the same root: `Z(A | B) = Z(A) && Z(B)`, so a common root reduces by idempotence.
- Keep the existing `(X >> C) or/add zext((X & lowmask(C)) != 0)` handling as a separate root source.

This is intentionally not an `or(ctpop(X), X)`-specific fold. The OR logic does not know about any intrinsic names; it only compares the zero-test roots of both operands.

Scope:
- This patch does not introduce a general boolean expression solver.
- It does not model lane-reordering or reduction intrinsics.
- It keeps the first revision limited to `ctpop`, `bswap`, `bitreverse`, and common-root OR handling.

Correctness sketch:
For each supported intrinsic, per scalar lane:

```
ctpop(x) == 0      iff x == 0
bswap(x) == 0      iff x == 0
bitreverse(x) == 0 iff x == 0
```

For bitwise OR, per scalar lane:

```
(a | b) == 0 iff a == 0 && b == 0
```

If both OR operands reduce to the same zero-test root `x`, then:

```
(A | B) == 0
iff A == 0 && B == 0
iff x == 0 && x == 0
iff x == 0
```

The vector cases follow lane-wise. The helper comment explicitly requires per-lane zero-test equivalence, which is stronger than aggregate non-zeroness.

Tests:
- `ninja -C /home/work/llama_young/llvm-zero-test-root-build opt FileCheck count not llvm-config`
- `llvm/utils/update_test_checks.py --opt-binary /home/work/llama_young/llvm-zero-test-root-build/bin/opt llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll`
- `/home/work/llama_young/llvm-zero-test-root-build/bin/opt -passes=instcombine -S < llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll | /home/work/llama_young/llvm-zero-test-root-build/bin/FileCheck llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll`
- `/home/work/llama_young/llvm-zero-test-root-build/bin/llvm-lit -sv llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll`
- `git diff --check`

---
Full diff: https://github.com/llvm/llvm-project/pull/208379.diff


2 Files Affected:

- (modified) llvm/lib/Analysis/ValueTracking.cpp (+97-18) 
- (added) llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll (+115) 


``````````diff
diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index 59631873305d4..910c8b7bce199 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -308,6 +308,7 @@ bool llvm::isKnownToBeAPowerOfTwo(const Value *V, const DataLayout &DL,
 
 static bool isKnownNonZero(const Value *V, const APInt &DemandedElts,
                            const SimplifyQuery &Q, unsigned Depth);
+static const Value *getOperandWithSameZeroTest(const Value *V);
 
 bool llvm::isKnownNonNegative(const Value *V, const SimplifyQuery &SQ,
                               unsigned Depth) {
@@ -3605,13 +3606,13 @@ static bool isKnownNonZeroFromOperator(const Operator *I,
     }
 
     if (auto *II = dyn_cast<IntrinsicInst>(I)) {
+      if (const Value *Op = getOperandWithSameZeroTest(II))
+        return isKnownNonZero(Op, DemandedElts, Q, Depth);
+
       switch (II->getIntrinsicID()) {
       case Intrinsic::sshl_sat:
       case Intrinsic::ushl_sat:
       case Intrinsic::abs:
-      case Intrinsic::bitreverse:
-      case Intrinsic::bswap:
-      case Intrinsic::ctpop:
         return isKnownNonZero(II->getArgOperand(0), DemandedElts, Q, Depth);
         // NB: We don't do usub_sat here as in any case we can prove its
         // non-zero, we will fold it to `sub nuw` in InstCombine.
@@ -10744,25 +10745,103 @@ void llvm::findValuesAffectedByCondition(
   }
 }
 
-const Value *llvm::stripNullTest(const Value *V) {
-  // (X >> C) or/add (X & mask(C) != 0)
-  if (const auto *BO = dyn_cast<BinaryOperator>(V)) {
-    if (BO->getOpcode() == Instruction::Add ||
-        BO->getOpcode() == Instruction::Or) {
-      const Value *X;
-      const APInt *C1, *C2;
-      if (match(BO, m_c_BinOp(m_LShr(m_Value(X), m_APInt(C1)),
-                              m_ZExt(m_SpecificICmp(
-                                  ICmpInst::ICMP_NE,
-                                  m_And(m_Deferred(X), m_LowBitMask(C2)),
-                                  m_Zero())))) &&
-          C2->popcount() == C1->getZExtValue())
-        return X;
-    }
+static const Value *getOperandWithSameZeroTest(const Value *V) {
+  const auto *II = dyn_cast<IntrinsicInst>(V);
+  if (!II)
+    return nullptr;
+
+  switch (II->getIntrinsicID()) {
+  case Intrinsic::bitreverse:
+  case Intrinsic::bswap:
+  case Intrinsic::ctpop:
+    // Return only operands with the same per-lane zero-test as the original
+    // value:
+    //
+    //   (Op X) == 0  iff  X == 0
+    //
+    // This is intentionally stronger than preserving aggregate non-zeroness.
+    // For example, a lane-reordering intrinsic may preserve whether an entire
+    // vector is non-zero, but it would not preserve the per-lane result of
+    // `icmp eq/ne V, zeroinitializer`. Keep this list limited to operations
+    // that preserve the zero-test independently in every scalar lane.
+    return II->getArgOperand(0);
+  default:
+    return nullptr;
   }
+}
+
+static const Value *stripExistingNullTestPattern(const Value *V) {
+  // This preserves the historical stripNullTest fold:
+  //
+  //   (X >> C) or/add zext((X & lowmask(C)) != 0)
+  //
+  // The expression is zero exactly when X is zero: the shift covers the high
+  // bits and the low-mask test covers the bits shifted out. Keep this as a
+  // separate helper so the zero-test-root logic below can use it as another
+  // source of root equivalence without mixing it into the OR algebra.
+  const auto *BO = dyn_cast<BinaryOperator>(V);
+  if (!BO || (BO->getOpcode() != Instruction::Add &&
+              BO->getOpcode() != Instruction::Or))
+    return nullptr;
+
+  const Value *X;
+  const APInt *C1, *C2;
+  if (match(BO, m_c_BinOp(m_LShr(m_Value(X), m_APInt(C1)),
+                          m_ZExt(m_SpecificICmp(
+                              ICmpInst::ICMP_NE,
+                              m_And(m_Deferred(X), m_LowBitMask(C2)),
+                              m_Zero())))) &&
+      C2->popcount() == C1->getZExtValue())
+    return X;
+
   return nullptr;
 }
 
+static const Value *getZeroTestRoot(const Value *V, unsigned Depth = 0) {
+  if (Depth >= MaxAnalysisRecursionDepth)
+    return V;
+
+  // First peel operations whose result has the exact same zero-test as one
+  // operand. This step does not inspect how the value is used; it only records
+  // a local equivalence for `V == 0`.
+  if (const Value *X = getOperandWithSameZeroTest(V))
+    return getZeroTestRoot(X, Depth + 1);
+
+  // Then preserve existing stripNullTest knowledge as another zero-test root.
+  // Recurse so a future root can still be stripped through a chain of such
+  // expressions without duplicating the caller-side logic.
+  if (const Value *X = stripExistingNullTestPattern(V))
+    return getZeroTestRoot(X, Depth + 1);
+
+  const auto *BO = dyn_cast<BinaryOperator>(V);
+  if (!BO || BO->getOpcode() != Instruction::Or)
+    return V;
+
+  const Value *LHS = getZeroTestRoot(BO->getOperand(0), Depth + 1);
+  const Value *RHS = getZeroTestRoot(BO->getOperand(1), Depth + 1);
+
+  // The only expression-level algebra modeled here is bitwise OR:
+  //
+  //   Z(A | B) = Z(A) && Z(B), where Z(V) is `V == 0`.
+  //
+  // If both operands reduce to the same root X, the conjunction is idempotent:
+  //
+  //   Z(A | B) = Z(X) && Z(X) = Z(X)
+  //
+  // If the roots differ, this helper deliberately gives up. That keeps this a
+  // small canonicalization for common zero-test roots, not a general boolean
+  // expression solver.
+  if (LHS == RHS)
+    return LHS;
+
+  return V;
+}
+
+const Value *llvm::stripNullTest(const Value *V) {
+  const Value *Root = getZeroTestRoot(V);
+  return Root != V ? Root : nullptr;
+}
+
 Value *llvm::stripNullTest(Value *V) {
   return const_cast<Value *>(stripNullTest(const_cast<const Value *>(V)));
 }
diff --git a/llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll b/llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll
new file mode 100644
index 0000000000000..b0f812d3297a9
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll
@@ -0,0 +1,115 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=instcombine -S < %s | FileCheck %s
+
+declare i8 @llvm.ctpop.i8(i8)
+declare i32 @llvm.bitreverse.i32(i32)
+declare i32 @llvm.bswap.i32(i32)
+declare i32 @llvm.ctpop.i32(i32)
+declare <2 x i8> @llvm.bitreverse.v2i8(<2 x i8>)
+declare <2 x i8> @llvm.ctpop.v2i8(<2 x i8>)
+
+define i1 @ctpop_zero_test(i8 %x) {
+; CHECK-LABEL: define i1 @ctpop_zero_test(
+; CHECK-SAME: i8 [[X:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i8 [[X]], 0
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+  %p = call i8 @llvm.ctpop.i8(i8 %x)
+  %cmp = icmp eq i8 %p, 0
+  ret i1 %cmp
+}
+
+define i1 @or_ctpop_self_ne_zero(i32 %x) {
+; CHECK-LABEL: define i1 @or_ctpop_self_ne_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = icmp ne i32 [[X]], 0
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %p = call i32 @llvm.ctpop.i32(i32 %x)
+  %o = or i32 %p, %x
+  %r = icmp ne i32 %o, 0
+  ret i1 %r
+}
+
+define i1 @or_ctpop_self_commuted_ne_zero(i32 %x) {
+; CHECK-LABEL: define i1 @or_ctpop_self_commuted_ne_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = icmp ne i32 [[X]], 0
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %p = call i32 @llvm.ctpop.i32(i32 %x)
+  %o = or i32 %x, %p
+  %r = icmp ne i32 %o, 0
+  ret i1 %r
+}
+
+define i1 @or_ctpop_bswap_ne_zero(i32 %x) {
+; CHECK-LABEL: define i1 @or_ctpop_bswap_ne_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = icmp ne i32 [[X]], 0
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %p = call i32 @llvm.ctpop.i32(i32 %x)
+  %s = call i32 @llvm.bswap.i32(i32 %x)
+  %o = or i32 %p, %s
+  %r = icmp ne i32 %o, 0
+  ret i1 %r
+}
+
+define i1 @or_bitreverse_bswap_eq_zero(i32 %x) {
+; CHECK-LABEL: define i1 @or_bitreverse_bswap_eq_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = icmp eq i32 [[X]], 0
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %r = call i32 @llvm.bitreverse.i32(i32 %x)
+  %s = call i32 @llvm.bswap.i32(i32 %x)
+  %o = or i32 %r, %s
+  %cmp = icmp eq i32 %o, 0
+  ret i1 %cmp
+}
+
+define i1 @nested_or_zero_test_root(i32 %x) {
+; CHECK-LABEL: define i1 @nested_or_zero_test_root(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i32 [[X]], 0
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+  %p = call i32 @llvm.ctpop.i32(i32 %x)
+  %r = call i32 @llvm.bitreverse.i32(i32 %x)
+  %s = call i32 @llvm.bswap.i32(i32 %x)
+  %o1 = or i32 %p, %x
+  %o2 = or i32 %o1, %r
+  %o3 = or i32 %o2, %s
+  %cmp = icmp eq i32 %o3, 0
+  ret i1 %cmp
+}
+
+define <2 x i1> @or_ctpop_bitreverse_vec_ne_zero(<2 x i8> %x) {
+; CHECK-LABEL: define <2 x i1> @or_ctpop_bitreverse_vec_ne_zero(
+; CHECK-SAME: <2 x i8> [[X:%.*]]) {
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne <2 x i8> [[X]], zeroinitializer
+; CHECK-NEXT:    ret <2 x i1> [[CMP]]
+;
+  %p = call <2 x i8> @llvm.ctpop.v2i8(<2 x i8> %x)
+  %r = call <2 x i8> @llvm.bitreverse.v2i8(<2 x i8> %x)
+  %o = or <2 x i8> %p, %r
+  %cmp = icmp ne <2 x i8> %o, zeroinitializer
+  ret <2 x i1> %cmp
+}
+
+define i1 @or_ctpop_x_bswap_y_ne_zero(i32 %x, i32 %y) {
+; CHECK-LABEL: define i1 @or_ctpop_x_bswap_y_ne_zero(
+; CHECK-SAME: i32 [[X:%.*]], i32 [[Y:%.*]]) {
+; CHECK-NEXT:    [[P:%.*]] = call range(i32 0, 33) i32 @llvm.ctpop.i32(i32 [[X]])
+; CHECK-NEXT:    [[S:%.*]] = call i32 @llvm.bswap.i32(i32 [[Y]])
+; CHECK-NEXT:    [[O:%.*]] = or i32 [[P]], [[S]]
+; CHECK-NEXT:    [[R:%.*]] = icmp ne i32 [[O]], 0
+; CHECK-NEXT:    ret i1 [[R]]
+;
+  %p = call i32 @llvm.ctpop.i32(i32 %x)
+  %s = call i32 @llvm.bswap.i32(i32 %y)
+  %o = or i32 %p, %s
+  %r = icmp ne i32 %o, 0
+  ret i1 %r
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/208379


More information about the llvm-commits mailing list