[llvm] [ValueTracking] Strip zero tests through common zero-test roots (PR #208379)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Jul 8 22:22:44 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-analysis
Author: Jaeuk Lee (skku970412)
<details>
<summary>Changes</summary>
Summary:
- Rework `stripNullTest` around a small zero-test-root canonicalization.
- Add a shared `getOperandWithSameZeroTest()` helper for lane-preserving intrinsics whose per-lane zero-test is equivalent to operand 0: `ctpop`, `bswap`, and `bitreverse`.
- Reuse that helper in `isKnownNonZero()` for the same intrinsic knowledge.
- Model the OR zero-test rule only when both sides reduce to the same root: `Z(A | B) = Z(A) && Z(B)`, so a common root reduces by idempotence.
- Keep the existing `(X >> C) or/add zext((X & lowmask(C)) != 0)` handling as a separate root source.
This is intentionally not an `or(ctpop(X), X)`-specific fold. The OR logic does not know about any intrinsic names; it only compares the zero-test roots of both operands.
Scope:
- This patch does not introduce a general boolean expression solver.
- It does not model lane-reordering or reduction intrinsics.
- It keeps the first revision limited to `ctpop`, `bswap`, `bitreverse`, and common-root OR handling.
Correctness sketch:
For each supported intrinsic, per scalar lane:
```
ctpop(x) == 0 iff x == 0
bswap(x) == 0 iff x == 0
bitreverse(x) == 0 iff x == 0
```
For bitwise OR, per scalar lane:
```
(a | b) == 0 iff a == 0 && b == 0
```
If both OR operands reduce to the same zero-test root `x`, then:
```
(A | B) == 0
iff A == 0 && B == 0
iff x == 0 && x == 0
iff x == 0
```
The vector cases follow lane-wise. The helper comment explicitly requires per-lane zero-test equivalence, which is stronger than aggregate non-zeroness.
Tests:
- `ninja -C /home/work/llama_young/llvm-zero-test-root-build opt FileCheck count not llvm-config`
- `llvm/utils/update_test_checks.py --opt-binary /home/work/llama_young/llvm-zero-test-root-build/bin/opt llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll`
- `/home/work/llama_young/llvm-zero-test-root-build/bin/opt -passes=instcombine -S < llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll | /home/work/llama_young/llvm-zero-test-root-build/bin/FileCheck llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll`
- `/home/work/llama_young/llvm-zero-test-root-build/bin/llvm-lit -sv llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll`
- `git diff --check`
---
Full diff: https://github.com/llvm/llvm-project/pull/208379.diff
2 Files Affected:
- (modified) llvm/lib/Analysis/ValueTracking.cpp (+97-18)
- (added) llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll (+115)
``````````diff
diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index 59631873305d4..910c8b7bce199 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -308,6 +308,7 @@ bool llvm::isKnownToBeAPowerOfTwo(const Value *V, const DataLayout &DL,
static bool isKnownNonZero(const Value *V, const APInt &DemandedElts,
const SimplifyQuery &Q, unsigned Depth);
+static const Value *getOperandWithSameZeroTest(const Value *V);
bool llvm::isKnownNonNegative(const Value *V, const SimplifyQuery &SQ,
unsigned Depth) {
@@ -3605,13 +3606,13 @@ static bool isKnownNonZeroFromOperator(const Operator *I,
}
if (auto *II = dyn_cast<IntrinsicInst>(I)) {
+ if (const Value *Op = getOperandWithSameZeroTest(II))
+ return isKnownNonZero(Op, DemandedElts, Q, Depth);
+
switch (II->getIntrinsicID()) {
case Intrinsic::sshl_sat:
case Intrinsic::ushl_sat:
case Intrinsic::abs:
- case Intrinsic::bitreverse:
- case Intrinsic::bswap:
- case Intrinsic::ctpop:
return isKnownNonZero(II->getArgOperand(0), DemandedElts, Q, Depth);
// NB: We don't do usub_sat here as in any case we can prove its
// non-zero, we will fold it to `sub nuw` in InstCombine.
@@ -10744,25 +10745,103 @@ void llvm::findValuesAffectedByCondition(
}
}
-const Value *llvm::stripNullTest(const Value *V) {
- // (X >> C) or/add (X & mask(C) != 0)
- if (const auto *BO = dyn_cast<BinaryOperator>(V)) {
- if (BO->getOpcode() == Instruction::Add ||
- BO->getOpcode() == Instruction::Or) {
- const Value *X;
- const APInt *C1, *C2;
- if (match(BO, m_c_BinOp(m_LShr(m_Value(X), m_APInt(C1)),
- m_ZExt(m_SpecificICmp(
- ICmpInst::ICMP_NE,
- m_And(m_Deferred(X), m_LowBitMask(C2)),
- m_Zero())))) &&
- C2->popcount() == C1->getZExtValue())
- return X;
- }
+static const Value *getOperandWithSameZeroTest(const Value *V) {
+ const auto *II = dyn_cast<IntrinsicInst>(V);
+ if (!II)
+ return nullptr;
+
+ switch (II->getIntrinsicID()) {
+ case Intrinsic::bitreverse:
+ case Intrinsic::bswap:
+ case Intrinsic::ctpop:
+ // Return only operands with the same per-lane zero-test as the original
+ // value:
+ //
+ // (Op X) == 0 iff X == 0
+ //
+ // This is intentionally stronger than preserving aggregate non-zeroness.
+ // For example, a lane-reordering intrinsic may preserve whether an entire
+ // vector is non-zero, but it would not preserve the per-lane result of
+ // `icmp eq/ne V, zeroinitializer`. Keep this list limited to operations
+ // that preserve the zero-test independently in every scalar lane.
+ return II->getArgOperand(0);
+ default:
+ return nullptr;
}
+}
+
+static const Value *stripExistingNullTestPattern(const Value *V) {
+ // This preserves the historical stripNullTest fold:
+ //
+ // (X >> C) or/add zext((X & lowmask(C)) != 0)
+ //
+ // The expression is zero exactly when X is zero: the shift covers the high
+ // bits and the low-mask test covers the bits shifted out. Keep this as a
+ // separate helper so the zero-test-root logic below can use it as another
+ // source of root equivalence without mixing it into the OR algebra.
+ const auto *BO = dyn_cast<BinaryOperator>(V);
+ if (!BO || (BO->getOpcode() != Instruction::Add &&
+ BO->getOpcode() != Instruction::Or))
+ return nullptr;
+
+ const Value *X;
+ const APInt *C1, *C2;
+ if (match(BO, m_c_BinOp(m_LShr(m_Value(X), m_APInt(C1)),
+ m_ZExt(m_SpecificICmp(
+ ICmpInst::ICMP_NE,
+ m_And(m_Deferred(X), m_LowBitMask(C2)),
+ m_Zero())))) &&
+ C2->popcount() == C1->getZExtValue())
+ return X;
+
return nullptr;
}
+static const Value *getZeroTestRoot(const Value *V, unsigned Depth = 0) {
+ if (Depth >= MaxAnalysisRecursionDepth)
+ return V;
+
+ // First peel operations whose result has the exact same zero-test as one
+ // operand. This step does not inspect how the value is used; it only records
+ // a local equivalence for `V == 0`.
+ if (const Value *X = getOperandWithSameZeroTest(V))
+ return getZeroTestRoot(X, Depth + 1);
+
+ // Then preserve existing stripNullTest knowledge as another zero-test root.
+ // Recurse so a future root can still be stripped through a chain of such
+ // expressions without duplicating the caller-side logic.
+ if (const Value *X = stripExistingNullTestPattern(V))
+ return getZeroTestRoot(X, Depth + 1);
+
+ const auto *BO = dyn_cast<BinaryOperator>(V);
+ if (!BO || BO->getOpcode() != Instruction::Or)
+ return V;
+
+ const Value *LHS = getZeroTestRoot(BO->getOperand(0), Depth + 1);
+ const Value *RHS = getZeroTestRoot(BO->getOperand(1), Depth + 1);
+
+ // The only expression-level algebra modeled here is bitwise OR:
+ //
+ // Z(A | B) = Z(A) && Z(B), where Z(V) is `V == 0`.
+ //
+ // If both operands reduce to the same root X, the conjunction is idempotent:
+ //
+ // Z(A | B) = Z(X) && Z(X) = Z(X)
+ //
+ // If the roots differ, this helper deliberately gives up. That keeps this a
+ // small canonicalization for common zero-test roots, not a general boolean
+ // expression solver.
+ if (LHS == RHS)
+ return LHS;
+
+ return V;
+}
+
+const Value *llvm::stripNullTest(const Value *V) {
+ const Value *Root = getZeroTestRoot(V);
+ return Root != V ? Root : nullptr;
+}
+
Value *llvm::stripNullTest(Value *V) {
return const_cast<Value *>(stripNullTest(const_cast<const Value *>(V)));
}
diff --git a/llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll b/llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll
new file mode 100644
index 0000000000000..b0f812d3297a9
--- /dev/null
+++ b/llvm/test/Transforms/InstCombine/icmp-zero-test-root.ll
@@ -0,0 +1,115 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=instcombine -S < %s | FileCheck %s
+
+declare i8 @llvm.ctpop.i8(i8)
+declare i32 @llvm.bitreverse.i32(i32)
+declare i32 @llvm.bswap.i32(i32)
+declare i32 @llvm.ctpop.i32(i32)
+declare <2 x i8> @llvm.bitreverse.v2i8(<2 x i8>)
+declare <2 x i8> @llvm.ctpop.v2i8(<2 x i8>)
+
+define i1 @ctpop_zero_test(i8 %x) {
+; CHECK-LABEL: define i1 @ctpop_zero_test(
+; CHECK-SAME: i8 [[X:%.*]]) {
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i8 [[X]], 0
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %p = call i8 @llvm.ctpop.i8(i8 %x)
+ %cmp = icmp eq i8 %p, 0
+ ret i1 %cmp
+}
+
+define i1 @or_ctpop_self_ne_zero(i32 %x) {
+; CHECK-LABEL: define i1 @or_ctpop_self_ne_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[R:%.*]] = icmp ne i32 [[X]], 0
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %p = call i32 @llvm.ctpop.i32(i32 %x)
+ %o = or i32 %p, %x
+ %r = icmp ne i32 %o, 0
+ ret i1 %r
+}
+
+define i1 @or_ctpop_self_commuted_ne_zero(i32 %x) {
+; CHECK-LABEL: define i1 @or_ctpop_self_commuted_ne_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[R:%.*]] = icmp ne i32 [[X]], 0
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %p = call i32 @llvm.ctpop.i32(i32 %x)
+ %o = or i32 %x, %p
+ %r = icmp ne i32 %o, 0
+ ret i1 %r
+}
+
+define i1 @or_ctpop_bswap_ne_zero(i32 %x) {
+; CHECK-LABEL: define i1 @or_ctpop_bswap_ne_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[R:%.*]] = icmp ne i32 [[X]], 0
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %p = call i32 @llvm.ctpop.i32(i32 %x)
+ %s = call i32 @llvm.bswap.i32(i32 %x)
+ %o = or i32 %p, %s
+ %r = icmp ne i32 %o, 0
+ ret i1 %r
+}
+
+define i1 @or_bitreverse_bswap_eq_zero(i32 %x) {
+; CHECK-LABEL: define i1 @or_bitreverse_bswap_eq_zero(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[R:%.*]] = icmp eq i32 [[X]], 0
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %r = call i32 @llvm.bitreverse.i32(i32 %x)
+ %s = call i32 @llvm.bswap.i32(i32 %x)
+ %o = or i32 %r, %s
+ %cmp = icmp eq i32 %o, 0
+ ret i1 %cmp
+}
+
+define i1 @nested_or_zero_test_root(i32 %x) {
+; CHECK-LABEL: define i1 @nested_or_zero_test_root(
+; CHECK-SAME: i32 [[X:%.*]]) {
+; CHECK-NEXT: [[CMP:%.*]] = icmp eq i32 [[X]], 0
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+ %p = call i32 @llvm.ctpop.i32(i32 %x)
+ %r = call i32 @llvm.bitreverse.i32(i32 %x)
+ %s = call i32 @llvm.bswap.i32(i32 %x)
+ %o1 = or i32 %p, %x
+ %o2 = or i32 %o1, %r
+ %o3 = or i32 %o2, %s
+ %cmp = icmp eq i32 %o3, 0
+ ret i1 %cmp
+}
+
+define <2 x i1> @or_ctpop_bitreverse_vec_ne_zero(<2 x i8> %x) {
+; CHECK-LABEL: define <2 x i1> @or_ctpop_bitreverse_vec_ne_zero(
+; CHECK-SAME: <2 x i8> [[X:%.*]]) {
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne <2 x i8> [[X]], zeroinitializer
+; CHECK-NEXT: ret <2 x i1> [[CMP]]
+;
+ %p = call <2 x i8> @llvm.ctpop.v2i8(<2 x i8> %x)
+ %r = call <2 x i8> @llvm.bitreverse.v2i8(<2 x i8> %x)
+ %o = or <2 x i8> %p, %r
+ %cmp = icmp ne <2 x i8> %o, zeroinitializer
+ ret <2 x i1> %cmp
+}
+
+define i1 @or_ctpop_x_bswap_y_ne_zero(i32 %x, i32 %y) {
+; CHECK-LABEL: define i1 @or_ctpop_x_bswap_y_ne_zero(
+; CHECK-SAME: i32 [[X:%.*]], i32 [[Y:%.*]]) {
+; CHECK-NEXT: [[P:%.*]] = call range(i32 0, 33) i32 @llvm.ctpop.i32(i32 [[X]])
+; CHECK-NEXT: [[S:%.*]] = call i32 @llvm.bswap.i32(i32 [[Y]])
+; CHECK-NEXT: [[O:%.*]] = or i32 [[P]], [[S]]
+; CHECK-NEXT: [[R:%.*]] = icmp ne i32 [[O]], 0
+; CHECK-NEXT: ret i1 [[R]]
+;
+ %p = call i32 @llvm.ctpop.i32(i32 %x)
+ %s = call i32 @llvm.bswap.i32(i32 %y)
+ %o = or i32 %p, %s
+ %r = icmp ne i32 %o, 0
+ ret i1 %r
+}
``````````
</details>
https://github.com/llvm/llvm-project/pull/208379
More information about the llvm-commits
mailing list