[llvm] [ValueTracking] Compute known bits of or/and/xor recurrences from start and step (PR #222334)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 9 06:58:17 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Konstantin Bogdanov (thevar1able)

<details>
<summary>Changes</summary>

For simple phi recurrences of the form `%iv = L op R` we currently only derive trailing zero bits. This adds full known-bits propagation for the bitwise operations, which subsumes the trailing-zeros rule for them:

* or/xor: bits that are zero in both the start value and the step stay zero. For or, bits that are one in the start value additionally stay one.
* and: bits that are zero in the start value stay zero, and bits that are one in both the start value and the step stay one.

The step only contributes from the second iteration on, so every fact must also hold for the start value alone.

This is needed to recognise that an or-reduction of zext'd i8 values fits in i8 once InstCombine has folded the zext/trunc round trip of a promoted accumulator into the phi (#<!-- -->222142). The vectorizer side is in #<!-- -->222183, which depends on this.

The two AMDGPU tests are adjusted to use an opaque start value for their recurrences, so that the improved known bits don't fold away the patterns they were written to test.


---
Full diff: https://github.com/llvm/llvm-project/pull/222334.diff


4 Files Affected:

- (modified) llvm/lib/Analysis/ValueTracking.cpp (+12-2) 
- (modified) llvm/test/CodeGen/AMDGPU/GlobalISel/is-safe-to-sink-bug.ll (+10-11) 
- (modified) llvm/test/CodeGen/AMDGPU/absdiff.ll (+2-3) 
- (modified) llvm/test/Transforms/InstCombine/recurrence.ll (+198) 


``````````diff
diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index 950b125028b99..e41d2f5bb2ba1 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -1882,6 +1882,7 @@ static void computeKnownBitsFromOperator(const Operator *I,
       case Instruction::Sub:
       case Instruction::And:
       case Instruction::Or:
+      case Instruction::Xor:
       case Instruction::Mul: {
         // Change the context instruction to the "edge" that flows into the
         // phi. This is important because that is where the value is actually
@@ -1903,8 +1904,17 @@ static void computeKnownBitsFromOperator(const Operator *I,
         RecQ.CxtI = LInst;
         computeKnownBits(L, DemandedElts, Known3, RecQ, Depth + 1);
 
-        Known.Zero.setLowBits(std::min(Known2.countMinTrailingZeros(),
-                                       Known3.countMinTrailingZeros()));
+        if (Opcode == Instruction::Or || Opcode == Instruction::Xor) {
+          Known.Zero |= Known2.Zero & Known3.Zero;
+          if (Opcode == Instruction::Or)
+            Known.One |= Known2.One;
+        } else if (Opcode == Instruction::And) {
+          Known.Zero |= Known2.Zero;
+          Known.One |= Known2.One & Known3.One;
+        } else {
+          Known.Zero.setLowBits(std::min(Known2.countMinTrailingZeros(),
+                                         Known3.countMinTrailingZeros()));
+        }
 
         auto *OverflowOp = dyn_cast<OverflowingBinaryOperator>(BO);
         if (!OverflowOp || !Q.IIQ.hasNoSignedWrap(OverflowOp))
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/is-safe-to-sink-bug.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/is-safe-to-sink-bug.ll
index d039334ab63dc..a1ea35304f113 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/is-safe-to-sink-bug.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/is-safe-to-sink-bug.ll
@@ -19,11 +19,11 @@ define amdgpu_ps void @_amdgpu_ps_main(i1 %arg) {
 ; CHECK-NEXT:    s_mov_b32 s2, s0
 ; CHECK-NEXT:    s_mov_b32 s3, s0
 ; CHECK-NEXT:    s_addc_u32 s9, s9, 0
-; CHECK-NEXT:    s_buffer_load_dword s1, s[0:3], 0x0
+; CHECK-NEXT:    s_buffer_load_dword s2, s[0:3], 0x0
 ; CHECK-NEXT:    s_mov_b32 s32, 0
 ; CHECK-NEXT:    v_cmp_ne_u32_e32 vcc_lo, 0, v0
 ; CHECK-NEXT:    s_waitcnt lgkmcnt(0)
-; CHECK-NEXT:    s_cmp_ge_i32 s1, 0
+; CHECK-NEXT:    s_cmp_ge_i32 s2, 0
 ; CHECK-NEXT:    s_cbranch_scc0 .LBB0_2
 ; CHECK-NEXT:  .LBB0_1: ; %bb12
 ; CHECK-NEXT:    v_cndmask_b32_e64 v0, 1.0, 0, s0
@@ -35,23 +35,22 @@ define amdgpu_ps void @_amdgpu_ps_main(i1 %arg) {
 ; CHECK-NEXT:    s_mov_b64 s[2:3], s[10:11]
 ; CHECK-NEXT:    s_swappc_b64 s[30:31], 0
 ; CHECK-NEXT:  .LBB0_2: ; %bb2.preheader
-; CHECK-NEXT:    s_mov_b32 s3, 0
 ; CHECK-NEXT:    s_mov_b32 s1, 0
-; CHECK-NEXT:    s_mov_b32 s2, 0
+; CHECK-NEXT:    s_mov_b32 s3, 0
 ; CHECK-NEXT:    s_branch .LBB0_4
 ; CHECK-NEXT:    .p2align 6
 ; CHECK-NEXT:  .LBB0_3: ; %bb6
 ; CHECK-NEXT:    ; in Loop: Header=BB0_4 Depth=1
 ; CHECK-NEXT:    s_or_b32 exec_lo, exec_lo, s0
 ; CHECK-NEXT:    v_cmp_ne_u32_e64 s0, 0, v0
-; CHECK-NEXT:    s_or_b32 s4, s3, 1
-; CHECK-NEXT:    s_and_b32 s0, s0, s2
-; CHECK-NEXT:    s_cmp_lt_i32 s3, 0
+; CHECK-NEXT:    s_or_b32 s4, s2, 1
+; CHECK-NEXT:    s_and_b32 s0, s0, s3
+; CHECK-NEXT:    s_cmp_lt_i32 s2, 0
 ; CHECK-NEXT:    s_cselect_b32 s5, 1, 0
-; CHECK-NEXT:    s_andn2_b32 s2, s2, exec_lo
+; CHECK-NEXT:    s_andn2_b32 s2, s3, exec_lo
 ; CHECK-NEXT:    s_and_b32 s3, exec_lo, s0
-; CHECK-NEXT:    s_or_b32 s2, s2, s3
-; CHECK-NEXT:    s_mov_b32 s3, s4
+; CHECK-NEXT:    s_or_b32 s3, s2, s3
+; CHECK-NEXT:    s_mov_b32 s2, s4
 ; CHECK-NEXT:    s_cmp_lg_u32 s5, 0
 ; CHECK-NEXT:    s_cbranch_scc0 .LBB0_1
 ; CHECK-NEXT:  .LBB0_4: ; %bb2
@@ -71,7 +70,7 @@ bb:
 
 bb2:
   %i3 = phi i1 [ %i9, %bb6 ], [ false, %bb ]
-  %i4 = phi i32 [ %i10, %bb6 ], [ 0, %bb ]
+  %i4 = phi i32 [ %i10, %bb6 ], [ %i, %bb ]
   br i1 %arg, label %bb5, label %bb6
 
 bb5:
diff --git a/llvm/test/CodeGen/AMDGPU/absdiff.ll b/llvm/test/CodeGen/AMDGPU/absdiff.ll
index e206b4341a096..6670459880ad2 100644
--- a/llvm/test/CodeGen/AMDGPU/absdiff.ll
+++ b/llvm/test/CodeGen/AMDGPU/absdiff.ll
@@ -2,10 +2,9 @@
 ; RUN: llc -mtriple=amdgpu9.00-amd-amdpal < %s | FileCheck %s
 
 
-define amdgpu_gs float @absdiff_valu_input_regression() {
+define amdgpu_gs float @absdiff_valu_input_regression(i32 inreg %start) {
 ; CHECK-LABEL: absdiff_valu_input_regression:
 ; CHECK:       ; %bb.0: ; %bb
-; CHECK-NEXT:    s_mov_b32 s0, 0
 ; CHECK-NEXT:  .LBB0_1: ; %bb1
 ; CHECK-NEXT:    ; =>This Inner Loop Header: Depth=1
 ; CHECK-NEXT:    s_mov_b32 s1, s0
@@ -22,7 +21,7 @@ bb:
   br label %bb1
 
 bb1:                                              ; preds = %bb1, %bb
-  %i = phi i32 [ 0, %bb ], [ %i9, %bb1 ]
+  %i = phi i32 [ %start, %bb ], [ %i9, %bb1 ]
   %i2 = phi i32 [ 0, %bb ], [ %i5, %bb1 ]
   %i3 = or i32 %i2, 1
   %i4 = or i32 %i3, 0
diff --git a/llvm/test/Transforms/InstCombine/recurrence.ll b/llvm/test/Transforms/InstCombine/recurrence.ll
index 6207009b531d5..3da3dc5e6f5a4 100644
--- a/llvm/test/Transforms/InstCombine/recurrence.ll
+++ b/llvm/test/Transforms/InstCombine/recurrence.ll
@@ -163,3 +163,201 @@ loop:                                             ; preds = %loop, %entry
 }
 
 declare void @use(i64)
+
+; The high bits of an or recurrence are known zero if they are zero in both
+; the start value and the step.
+define i32 @test_or_known_zero_high_bits(ptr %p, i64 %n) {
+; CHECK-LABEL: @test_or_known_zero_high_bits(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[ACC:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[ACC_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[P:%.*]], i64 [[IV]]
+; CHECK-NEXT:    [[X:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-NEXT:    [[X_EXT:%.*]] = zext i8 [[X]] to i32
+; CHECK-NEXT:    [[ACC_NEXT]] = or i32 [[ACC]], [[X_EXT]]
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N:%.*]]
+; CHECK-NEXT:    br i1 [[EC]], label [[EXIT:%.*]], label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i32 [[ACC_NEXT]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i32 [ 0, %entry ], [ %acc.next, %loop ]
+  %gep = getelementptr inbounds i8, ptr %p, i64 %iv
+  %x = load i8, ptr %gep
+  %x.ext = zext i8 %x to i32
+  %acc.next = or i32 %acc, %x.ext
+  %iv.next = add i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, %n
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  %masked = and i32 %acc.next, 255
+  ret i32 %masked
+}
+
+; Negative test: the start value has unknown high bits.
+define i32 @test_or_unknown_start(ptr %p, i64 %n, i32 %start) {
+; CHECK-LABEL: @test_or_unknown_start(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[ACC:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY]] ], [ [[ACC_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[P:%.*]], i64 [[IV]]
+; CHECK-NEXT:    [[X:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-NEXT:    [[X_EXT:%.*]] = zext i8 [[X]] to i32
+; CHECK-NEXT:    [[ACC_NEXT]] = or i32 [[ACC]], [[X_EXT]]
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N:%.*]]
+; CHECK-NEXT:    br i1 [[EC]], label [[EXIT:%.*]], label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    [[MASKED:%.*]] = and i32 [[ACC_NEXT]], 255
+; CHECK-NEXT:    ret i32 [[MASKED]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i32 [ %start, %entry ], [ %acc.next, %loop ]
+  %gep = getelementptr inbounds i8, ptr %p, i64 %iv
+  %x = load i8, ptr %gep
+  %x.ext = zext i8 %x to i32
+  %acc.next = or i32 %acc, %x.ext
+  %iv.next = add i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, %n
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  %masked = and i32 %acc.next, 255
+  ret i32 %masked
+}
+
+; Bits that are one in the start value stay one in an or recurrence, but bits
+; that are only one in the step do not apply to the first iteration.
+define i64 @test_or_known_one_start(i64 %a) {
+; CHECK-LABEL: @test_or_known_one_start(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    tail call void @use(i64 1)
+; CHECK-NEXT:    br label [[LOOP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 1, %entry ], [ %iv.next, %loop ]
+  %iv.next = or i64 %iv, %a
+  %bit0 = and i64 %iv, 1
+  tail call void @use(i64 %bit0)
+  br label %loop
+}
+
+; The high bits of an and recurrence are known zero if they are zero in the
+; start value.
+define i32 @test_and_known_zero_start(ptr %p, i64 %n) {
+; CHECK-LABEL: @test_and_known_zero_start(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[ACC:%.*]] = phi i32 [ 255, [[ENTRY]] ], [ [[ACC_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds [4 x i8], ptr [[P:%.*]], i64 [[IV]]
+; CHECK-NEXT:    [[X:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[ACC_NEXT]] = and i32 [[ACC]], [[X]]
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N:%.*]]
+; CHECK-NEXT:    br i1 [[EC]], label [[EXIT:%.*]], label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i32 [[ACC_NEXT]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i32 [ 255, %entry ], [ %acc.next, %loop ]
+  %gep = getelementptr inbounds i32, ptr %p, i64 %iv
+  %x = load i32, ptr %gep
+  %acc.next = and i32 %acc, %x
+  %iv.next = add i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, %n
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  %masked = and i32 %acc.next, 255
+  ret i32 %masked
+}
+
+; Negative test: an and recurrence does not inherit the zero bits of the step,
+; because the step does not apply to the first iteration.
+define i32 @test_and_step_zero_bits(i32 %start) {
+; CHECK-LABEL: @test_and_step_zero_bits(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    tail call void @use(i64 0)
+; CHECK-NEXT:    br i1 false, label [[LOOP]], label [[EXIT:%.*]]
+; CHECK:       exit:
+; CHECK-NEXT:    [[MASKED:%.*]] = and i32 [[START:%.*]], 4
+; CHECK-NEXT:    ret i32 [[MASKED]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %loop ]
+  %iv.next = and i32 %iv, 4
+  %masked = and i32 %iv, 4
+  tail call void @use(i64 0)
+  br i1 false, label %loop, label %exit
+
+exit:
+  ret i32 %masked
+}
+
+; The high bits of an xor recurrence are known zero if they are zero in both
+; the start value and the step.
+define i32 @test_xor_known_zero_high_bits(ptr %p, i64 %n) {
+; CHECK-LABEL: @test_xor_known_zero_high_bits(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[ACC:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[ACC_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i8, ptr [[P:%.*]], i64 [[IV]]
+; CHECK-NEXT:    [[X:%.*]] = load i8, ptr [[GEP]], align 1
+; CHECK-NEXT:    [[X_EXT:%.*]] = zext i8 [[X]] to i32
+; CHECK-NEXT:    [[ACC_NEXT]] = xor i32 [[ACC]], [[X_EXT]]
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i64 [[IV_NEXT]], [[N:%.*]]
+; CHECK-NEXT:    br i1 [[EC]], label [[EXIT:%.*]], label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i32 [[ACC_NEXT]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i32 [ 0, %entry ], [ %acc.next, %loop ]
+  %gep = getelementptr inbounds i8, ptr %p, i64 %iv
+  %x = load i8, ptr %gep
+  %x.ext = zext i8 %x to i32
+  %acc.next = xor i32 %acc, %x.ext
+  %iv.next = add i64 %iv, 1
+  %ec = icmp eq i64 %iv.next, %n
+  br i1 %ec, label %exit, label %loop
+
+exit:
+  %masked = and i32 %acc.next, 255
+  ret i32 %masked
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/222334


More information about the llvm-commits mailing list