[llvm] [AMDGPU] Fix amdgcn.mbcnt known bits conflicting with the range attribute (PR #227725)

Teja Alaghari via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 30 22:49:16 PDT 2026


https://github.com/TejaX-Alaghari updated https://github.com/llvm/llvm-project/pull/227725

>From f3db644a9f1e0e8532f25a1a23a6c83cf5aca851 Mon Sep 17 00:00:00 2001
From: tmahatej_amdeng <teja.alaghari at amd.com>
Date: Wed, 30 Sep 2026 18:53:43 +0530
Subject: [PATCH 1/4] Precommit test cases for mbcnt intrinsic calls being
 incorrectly folded to 1

---
 .../Transforms/InstCombine/AMDGPU/mbcnt.ll    | 49 +++++++++++++++++++
 1 file changed, 49 insertions(+)

diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
index 5f199131b177f..381ace30f28aa 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
@@ -292,6 +292,55 @@ define i32 @known_range_mbcnt_lo_refineable_range(i32 %unknown) {
   ret i32 %lo
 }
 
+; A zero mask contributes no lanes, so the call is the base. add and shl of that
+; base, including a variable shift amount, are the same replacement.
+define i32 @mbcnt_hi_zero_mask() {
+; DEFAULT-LABEL: define i32 @mbcnt_hi_zero_mask() {
+; DEFAULT-NEXT:    ret i32 1
+;
+; WAVE64-LABEL: define i32 @mbcnt_hi_zero_mask
+; WAVE64-SAME: () #[[ATTR1]] {
+; WAVE64-NEXT:    ret i32 1
+;
+  %hi = call i32 @llvm.amdgcn.mbcnt.hi(i32 0, i32 47666)
+  %xor = xor i32 %hi, 1
+  ret i32 %xor
+}
+
+; A non-zero mask is lane-varying, so the call must not become a constant.
+; mbcnt.hi is the same known-bits path; on wave32 it is separately a copy of
+; the base. shl is the same consumer as xor once the call is not known zero.
+define i32 @mbcnt_lo_nonzero_mask() {
+; DEFAULT-LABEL: define i32 @mbcnt_lo_nonzero_mask() {
+; DEFAULT-NEXT:    ret i32 1
+;
+; WAVE64-LABEL: define i32 @mbcnt_lo_nonzero_mask
+; WAVE64-SAME: () #[[ATTR1]] {
+; WAVE64-NEXT:    ret i32 1
+;
+  %lo = call i32 @llvm.amdgcn.mbcnt.lo(i32 48938, i32 48938)
+  %xor = xor i32 %lo, 1
+  ret i32 %xor
+}
+
+; The mask load is live. Different masks must not fold the two calls together.
+define i32 @mbcnt_hi_load_mask(ptr addrspace(1) %in) {
+; DEFAULT-LABEL: define i32 @mbcnt_hi_load_mask
+; DEFAULT-SAME: (ptr addrspace(1) [[IN:%.*]]) {
+; DEFAULT-NEXT:    ret i32 1
+;
+; WAVE64-LABEL: define i32 @mbcnt_hi_load_mask
+; WAVE64-SAME: (ptr addrspace(1) [[IN:%.*]]) #[[ATTR1]] {
+; WAVE64-NEXT:    ret i32 1
+;
+  %mask = load i32, ptr addrspace(1) %in, align 4
+  %hi0 = call i32 @llvm.amdgcn.mbcnt.hi(i32 793772030, i32 793772030)
+  %hi1 = call i32 @llvm.amdgcn.mbcnt.hi(i32 %mask, i32 793772030)
+  %xor = xor i32 %hi0, %hi1
+  %mix = xor i32 %xor, 1
+  ret i32 %mix
+}
+
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; CHECK: {{.*}}
 ; WAVE32: {{.*}}

>From 9152ed63cb3b5b71e48c8e6cd99a3b7f7e167e72 Mon Sep 17 00:00:00 2001
From: tmahatej_amdeng <teja.alaghari at amd.com>
Date: Wed, 30 Sep 2026 19:06:09 +0530
Subject: [PATCH 2/4] Remove confusing comments about other add/shfl cases

---
 llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll | 5 +----
 1 file changed, 1 insertion(+), 4 deletions(-)

diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
index 381ace30f28aa..38c9816684805 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
@@ -292,8 +292,7 @@ define i32 @known_range_mbcnt_lo_refineable_range(i32 %unknown) {
   ret i32 %lo
 }
 
-; A zero mask contributes no lanes, so the call is the base. add and shl of that
-; base, including a variable shift amount, are the same replacement.
+; A zero mask contributes no lanes, so the call is the base.
 define i32 @mbcnt_hi_zero_mask() {
 ; DEFAULT-LABEL: define i32 @mbcnt_hi_zero_mask() {
 ; DEFAULT-NEXT:    ret i32 1
@@ -308,8 +307,6 @@ define i32 @mbcnt_hi_zero_mask() {
 }
 
 ; A non-zero mask is lane-varying, so the call must not become a constant.
-; mbcnt.hi is the same known-bits path; on wave32 it is separately a copy of
-; the base. shl is the same consumer as xor once the call is not known zero.
 define i32 @mbcnt_lo_nonzero_mask() {
 ; DEFAULT-LABEL: define i32 @mbcnt_lo_nonzero_mask() {
 ; DEFAULT-NEXT:    ret i32 1

>From 7870b69401f293563449a57cfc7850617910cdf9 Mon Sep 17 00:00:00 2001
From: tmahatej_amdeng <teja.alaghari at amd.com>
Date: Wed, 30 Sep 2026 18:58:00 +0530
Subject: [PATCH 3/4] [AMDGPU] Fix amdgcn.mbcnt known bits conflicting with the
 range attribute

---
 llvm/lib/Analysis/ValueTracking.cpp           |  5 ++--
 .../AMDGPU/AMDGPUInstCombineIntrinsic.cpp     |  4 +++
 .../Transforms/InstCombine/AMDGPU/mbcnt.ll    | 26 ++++++++++++++-----
 3 files changed, 27 insertions(+), 8 deletions(-)

diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index f56375e4dbe73..0dc5d03c91381 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -2374,10 +2374,11 @@ static void computeKnownBitsFromOperator(const Operator *I,
       case Intrinsic::amdgcn_mbcnt_lo: {
         // Wave64 mbcnt_lo returns at most 32 + src1. Otherwise these return at
         // most 31 + src1.
-        Known.Zero.setBitsFrom(
+        KnownBits MbcntKnown(BitWidth);
+        MbcntKnown.Zero.setBitsFrom(
             II->getIntrinsicID() == Intrinsic::amdgcn_mbcnt_lo ? 6 : 5);
         computeKnownBits(I->getOperand(1), Known2, Q, Depth + 1);
-        Known = KnownBits::add(Known, Known2);
+        Known = Known.unionWith(KnownBits::add(MbcntKnown, Known2));
         break;
       }
       case Intrinsic::vscale: {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
index 4936b16148ce4..b29496a597684 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
@@ -1744,6 +1744,10 @@ GCNTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
       return IC.replaceInstUsesWith(II, II.getArgOperand(1));
     [[fallthrough]];
   case Intrinsic::amdgcn_mbcnt_lo: {
+    // No lanes contribute when the mask is zero.
+    if (match(II.getArgOperand(0), m_Zero()))
+      return IC.replaceInstUsesWith(II, II.getArgOperand(1));
+
     ConstantRange AccRange =
         computeConstantRange(II.getArgOperand(1),
                              /*ForSigned=*/false, IC.getSimplifyQuery());
diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
index 38c9816684805..453f9c305de1a 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
@@ -295,11 +295,11 @@ define i32 @known_range_mbcnt_lo_refineable_range(i32 %unknown) {
 ; A zero mask contributes no lanes, so the call is the base.
 define i32 @mbcnt_hi_zero_mask() {
 ; DEFAULT-LABEL: define i32 @mbcnt_hi_zero_mask() {
-; DEFAULT-NEXT:    ret i32 1
+; DEFAULT-NEXT:    ret i32 47667
 ;
 ; WAVE64-LABEL: define i32 @mbcnt_hi_zero_mask
 ; WAVE64-SAME: () #[[ATTR1]] {
-; WAVE64-NEXT:    ret i32 1
+; WAVE64-NEXT:    ret i32 47667
 ;
   %hi = call i32 @llvm.amdgcn.mbcnt.hi(i32 0, i32 47666)
   %xor = xor i32 %hi, 1
@@ -309,11 +309,15 @@ define i32 @mbcnt_hi_zero_mask() {
 ; A non-zero mask is lane-varying, so the call must not become a constant.
 define i32 @mbcnt_lo_nonzero_mask() {
 ; DEFAULT-LABEL: define i32 @mbcnt_lo_nonzero_mask() {
-; DEFAULT-NEXT:    ret i32 1
+; DEFAULT-NEXT:    [[LO:%.*]] = call range(i32 48938, 48971) i32 @llvm.amdgcn.mbcnt.lo(i32 48938, i32 48938)
+; DEFAULT-NEXT:    [[XOR:%.*]] = xor i32 [[LO]], 1
+; DEFAULT-NEXT:    ret i32 [[XOR]]
 ;
 ; WAVE64-LABEL: define i32 @mbcnt_lo_nonzero_mask
 ; WAVE64-SAME: () #[[ATTR1]] {
-; WAVE64-NEXT:    ret i32 1
+; WAVE64-NEXT:    [[LO:%.*]] = call range(i32 48938, 48971) i32 @llvm.amdgcn.mbcnt.lo(i32 48938, i32 48938)
+; WAVE64-NEXT:    [[XOR:%.*]] = xor i32 [[LO]], 1
+; WAVE64-NEXT:    ret i32 [[XOR]]
 ;
   %lo = call i32 @llvm.amdgcn.mbcnt.lo(i32 48938, i32 48938)
   %xor = xor i32 %lo, 1
@@ -324,11 +328,21 @@ define i32 @mbcnt_lo_nonzero_mask() {
 define i32 @mbcnt_hi_load_mask(ptr addrspace(1) %in) {
 ; DEFAULT-LABEL: define i32 @mbcnt_hi_load_mask
 ; DEFAULT-SAME: (ptr addrspace(1) [[IN:%.*]]) {
-; DEFAULT-NEXT:    ret i32 1
+; DEFAULT-NEXT:    [[MASK:%.*]] = load i32, ptr addrspace(1) [[IN]], align 4
+; DEFAULT-NEXT:    [[HI0:%.*]] = call range(i32 793772030, 793772063) i32 @llvm.amdgcn.mbcnt.hi(i32 793772030, i32 793772030)
+; DEFAULT-NEXT:    [[HI1:%.*]] = call range(i32 793772030, 793772063) i32 @llvm.amdgcn.mbcnt.hi(i32 [[MASK]], i32 793772030)
+; DEFAULT-NEXT:    [[XOR:%.*]] = xor i32 [[HI0]], [[HI1]]
+; DEFAULT-NEXT:    [[MIX:%.*]] = xor i32 [[XOR]], 1
+; DEFAULT-NEXT:    ret i32 [[MIX]]
 ;
 ; WAVE64-LABEL: define i32 @mbcnt_hi_load_mask
 ; WAVE64-SAME: (ptr addrspace(1) [[IN:%.*]]) #[[ATTR1]] {
-; WAVE64-NEXT:    ret i32 1
+; WAVE64-NEXT:    [[MASK:%.*]] = load i32, ptr addrspace(1) [[IN]], align 4
+; WAVE64-NEXT:    [[HI0:%.*]] = call range(i32 793772030, 793772063) i32 @llvm.amdgcn.mbcnt.hi(i32 793772030, i32 793772030)
+; WAVE64-NEXT:    [[HI1:%.*]] = call range(i32 793772030, 793772063) i32 @llvm.amdgcn.mbcnt.hi(i32 [[MASK]], i32 793772030)
+; WAVE64-NEXT:    [[XOR:%.*]] = xor i32 [[HI0]], [[HI1]]
+; WAVE64-NEXT:    [[MIX:%.*]] = xor i32 [[XOR]], 1
+; WAVE64-NEXT:    ret i32 [[MIX]]
 ;
   %mask = load i32, ptr addrspace(1) %in, align 4
   %hi0 = call i32 @llvm.amdgcn.mbcnt.hi(i32 793772030, i32 793772030)

>From 15aa5fe56d84429533c8766429809d4ba43b4d9e Mon Sep 17 00:00:00 2001
From: tmahatej_amdeng <teja.alaghari at amd.com>
Date: Thu, 1 Oct 2026 10:19:10 +0530
Subject: [PATCH 4/4] Revert peephole optimization for folding a zero mask to
 base in mbcnt

---
 .../Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp   |  4 ----
 llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll   | 14 --------------
 2 files changed, 18 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
index b29496a597684..4936b16148ce4 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
@@ -1744,10 +1744,6 @@ GCNTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
       return IC.replaceInstUsesWith(II, II.getArgOperand(1));
     [[fallthrough]];
   case Intrinsic::amdgcn_mbcnt_lo: {
-    // No lanes contribute when the mask is zero.
-    if (match(II.getArgOperand(0), m_Zero()))
-      return IC.replaceInstUsesWith(II, II.getArgOperand(1));
-
     ConstantRange AccRange =
         computeConstantRange(II.getArgOperand(1),
                              /*ForSigned=*/false, IC.getSimplifyQuery());
diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
index 453f9c305de1a..d66f324031bd8 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
@@ -292,20 +292,6 @@ define i32 @known_range_mbcnt_lo_refineable_range(i32 %unknown) {
   ret i32 %lo
 }
 
-; A zero mask contributes no lanes, so the call is the base.
-define i32 @mbcnt_hi_zero_mask() {
-; DEFAULT-LABEL: define i32 @mbcnt_hi_zero_mask() {
-; DEFAULT-NEXT:    ret i32 47667
-;
-; WAVE64-LABEL: define i32 @mbcnt_hi_zero_mask
-; WAVE64-SAME: () #[[ATTR1]] {
-; WAVE64-NEXT:    ret i32 47667
-;
-  %hi = call i32 @llvm.amdgcn.mbcnt.hi(i32 0, i32 47666)
-  %xor = xor i32 %hi, 1
-  ret i32 %xor
-}
-
 ; A non-zero mask is lane-varying, so the call must not become a constant.
 define i32 @mbcnt_lo_nonzero_mask() {
 ; DEFAULT-LABEL: define i32 @mbcnt_lo_nonzero_mask() {



More information about the llvm-commits mailing list