[llvm] [AMDGPU] Fix amdgcn.mbcnt known bits conflicting with the range attribute (PR #227725)
Teja Alaghari via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 22:49:16 PDT 2026
https://github.com/TejaX-Alaghari updated https://github.com/llvm/llvm-project/pull/227725
>From f3db644a9f1e0e8532f25a1a23a6c83cf5aca851 Mon Sep 17 00:00:00 2001
From: tmahatej_amdeng <teja.alaghari at amd.com>
Date: Wed, 30 Sep 2026 18:53:43 +0530
Subject: [PATCH 1/4] Precommit test cases for mbcnt intrinsic calls being
incorrectly folded to 1
---
.../Transforms/InstCombine/AMDGPU/mbcnt.ll | 49 +++++++++++++++++++
1 file changed, 49 insertions(+)
diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
index 5f199131b177f..381ace30f28aa 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
@@ -292,6 +292,55 @@ define i32 @known_range_mbcnt_lo_refineable_range(i32 %unknown) {
ret i32 %lo
}
+; A zero mask contributes no lanes, so the call is the base. add and shl of that
+; base, including a variable shift amount, are the same replacement.
+define i32 @mbcnt_hi_zero_mask() {
+; DEFAULT-LABEL: define i32 @mbcnt_hi_zero_mask() {
+; DEFAULT-NEXT: ret i32 1
+;
+; WAVE64-LABEL: define i32 @mbcnt_hi_zero_mask
+; WAVE64-SAME: () #[[ATTR1]] {
+; WAVE64-NEXT: ret i32 1
+;
+ %hi = call i32 @llvm.amdgcn.mbcnt.hi(i32 0, i32 47666)
+ %xor = xor i32 %hi, 1
+ ret i32 %xor
+}
+
+; A non-zero mask is lane-varying, so the call must not become a constant.
+; mbcnt.hi is the same known-bits path; on wave32 it is separately a copy of
+; the base. shl is the same consumer as xor once the call is not known zero.
+define i32 @mbcnt_lo_nonzero_mask() {
+; DEFAULT-LABEL: define i32 @mbcnt_lo_nonzero_mask() {
+; DEFAULT-NEXT: ret i32 1
+;
+; WAVE64-LABEL: define i32 @mbcnt_lo_nonzero_mask
+; WAVE64-SAME: () #[[ATTR1]] {
+; WAVE64-NEXT: ret i32 1
+;
+ %lo = call i32 @llvm.amdgcn.mbcnt.lo(i32 48938, i32 48938)
+ %xor = xor i32 %lo, 1
+ ret i32 %xor
+}
+
+; The mask load is live. Different masks must not fold the two calls together.
+define i32 @mbcnt_hi_load_mask(ptr addrspace(1) %in) {
+; DEFAULT-LABEL: define i32 @mbcnt_hi_load_mask
+; DEFAULT-SAME: (ptr addrspace(1) [[IN:%.*]]) {
+; DEFAULT-NEXT: ret i32 1
+;
+; WAVE64-LABEL: define i32 @mbcnt_hi_load_mask
+; WAVE64-SAME: (ptr addrspace(1) [[IN:%.*]]) #[[ATTR1]] {
+; WAVE64-NEXT: ret i32 1
+;
+ %mask = load i32, ptr addrspace(1) %in, align 4
+ %hi0 = call i32 @llvm.amdgcn.mbcnt.hi(i32 793772030, i32 793772030)
+ %hi1 = call i32 @llvm.amdgcn.mbcnt.hi(i32 %mask, i32 793772030)
+ %xor = xor i32 %hi0, %hi1
+ %mix = xor i32 %xor, 1
+ ret i32 %mix
+}
+
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
; CHECK: {{.*}}
; WAVE32: {{.*}}
>From 9152ed63cb3b5b71e48c8e6cd99a3b7f7e167e72 Mon Sep 17 00:00:00 2001
From: tmahatej_amdeng <teja.alaghari at amd.com>
Date: Wed, 30 Sep 2026 19:06:09 +0530
Subject: [PATCH 2/4] Remove confusing comments about other add/shfl cases
---
llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll | 5 +----
1 file changed, 1 insertion(+), 4 deletions(-)
diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
index 381ace30f28aa..38c9816684805 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
@@ -292,8 +292,7 @@ define i32 @known_range_mbcnt_lo_refineable_range(i32 %unknown) {
ret i32 %lo
}
-; A zero mask contributes no lanes, so the call is the base. add and shl of that
-; base, including a variable shift amount, are the same replacement.
+; A zero mask contributes no lanes, so the call is the base.
define i32 @mbcnt_hi_zero_mask() {
; DEFAULT-LABEL: define i32 @mbcnt_hi_zero_mask() {
; DEFAULT-NEXT: ret i32 1
@@ -308,8 +307,6 @@ define i32 @mbcnt_hi_zero_mask() {
}
; A non-zero mask is lane-varying, so the call must not become a constant.
-; mbcnt.hi is the same known-bits path; on wave32 it is separately a copy of
-; the base. shl is the same consumer as xor once the call is not known zero.
define i32 @mbcnt_lo_nonzero_mask() {
; DEFAULT-LABEL: define i32 @mbcnt_lo_nonzero_mask() {
; DEFAULT-NEXT: ret i32 1
>From 7870b69401f293563449a57cfc7850617910cdf9 Mon Sep 17 00:00:00 2001
From: tmahatej_amdeng <teja.alaghari at amd.com>
Date: Wed, 30 Sep 2026 18:58:00 +0530
Subject: [PATCH 3/4] [AMDGPU] Fix amdgcn.mbcnt known bits conflicting with the
range attribute
---
llvm/lib/Analysis/ValueTracking.cpp | 5 ++--
.../AMDGPU/AMDGPUInstCombineIntrinsic.cpp | 4 +++
.../Transforms/InstCombine/AMDGPU/mbcnt.ll | 26 ++++++++++++++-----
3 files changed, 27 insertions(+), 8 deletions(-)
diff --git a/llvm/lib/Analysis/ValueTracking.cpp b/llvm/lib/Analysis/ValueTracking.cpp
index f56375e4dbe73..0dc5d03c91381 100644
--- a/llvm/lib/Analysis/ValueTracking.cpp
+++ b/llvm/lib/Analysis/ValueTracking.cpp
@@ -2374,10 +2374,11 @@ static void computeKnownBitsFromOperator(const Operator *I,
case Intrinsic::amdgcn_mbcnt_lo: {
// Wave64 mbcnt_lo returns at most 32 + src1. Otherwise these return at
// most 31 + src1.
- Known.Zero.setBitsFrom(
+ KnownBits MbcntKnown(BitWidth);
+ MbcntKnown.Zero.setBitsFrom(
II->getIntrinsicID() == Intrinsic::amdgcn_mbcnt_lo ? 6 : 5);
computeKnownBits(I->getOperand(1), Known2, Q, Depth + 1);
- Known = KnownBits::add(Known, Known2);
+ Known = Known.unionWith(KnownBits::add(MbcntKnown, Known2));
break;
}
case Intrinsic::vscale: {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
index 4936b16148ce4..b29496a597684 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
@@ -1744,6 +1744,10 @@ GCNTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
return IC.replaceInstUsesWith(II, II.getArgOperand(1));
[[fallthrough]];
case Intrinsic::amdgcn_mbcnt_lo: {
+ // No lanes contribute when the mask is zero.
+ if (match(II.getArgOperand(0), m_Zero()))
+ return IC.replaceInstUsesWith(II, II.getArgOperand(1));
+
ConstantRange AccRange =
computeConstantRange(II.getArgOperand(1),
/*ForSigned=*/false, IC.getSimplifyQuery());
diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
index 38c9816684805..453f9c305de1a 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
@@ -295,11 +295,11 @@ define i32 @known_range_mbcnt_lo_refineable_range(i32 %unknown) {
; A zero mask contributes no lanes, so the call is the base.
define i32 @mbcnt_hi_zero_mask() {
; DEFAULT-LABEL: define i32 @mbcnt_hi_zero_mask() {
-; DEFAULT-NEXT: ret i32 1
+; DEFAULT-NEXT: ret i32 47667
;
; WAVE64-LABEL: define i32 @mbcnt_hi_zero_mask
; WAVE64-SAME: () #[[ATTR1]] {
-; WAVE64-NEXT: ret i32 1
+; WAVE64-NEXT: ret i32 47667
;
%hi = call i32 @llvm.amdgcn.mbcnt.hi(i32 0, i32 47666)
%xor = xor i32 %hi, 1
@@ -309,11 +309,15 @@ define i32 @mbcnt_hi_zero_mask() {
; A non-zero mask is lane-varying, so the call must not become a constant.
define i32 @mbcnt_lo_nonzero_mask() {
; DEFAULT-LABEL: define i32 @mbcnt_lo_nonzero_mask() {
-; DEFAULT-NEXT: ret i32 1
+; DEFAULT-NEXT: [[LO:%.*]] = call range(i32 48938, 48971) i32 @llvm.amdgcn.mbcnt.lo(i32 48938, i32 48938)
+; DEFAULT-NEXT: [[XOR:%.*]] = xor i32 [[LO]], 1
+; DEFAULT-NEXT: ret i32 [[XOR]]
;
; WAVE64-LABEL: define i32 @mbcnt_lo_nonzero_mask
; WAVE64-SAME: () #[[ATTR1]] {
-; WAVE64-NEXT: ret i32 1
+; WAVE64-NEXT: [[LO:%.*]] = call range(i32 48938, 48971) i32 @llvm.amdgcn.mbcnt.lo(i32 48938, i32 48938)
+; WAVE64-NEXT: [[XOR:%.*]] = xor i32 [[LO]], 1
+; WAVE64-NEXT: ret i32 [[XOR]]
;
%lo = call i32 @llvm.amdgcn.mbcnt.lo(i32 48938, i32 48938)
%xor = xor i32 %lo, 1
@@ -324,11 +328,21 @@ define i32 @mbcnt_lo_nonzero_mask() {
define i32 @mbcnt_hi_load_mask(ptr addrspace(1) %in) {
; DEFAULT-LABEL: define i32 @mbcnt_hi_load_mask
; DEFAULT-SAME: (ptr addrspace(1) [[IN:%.*]]) {
-; DEFAULT-NEXT: ret i32 1
+; DEFAULT-NEXT: [[MASK:%.*]] = load i32, ptr addrspace(1) [[IN]], align 4
+; DEFAULT-NEXT: [[HI0:%.*]] = call range(i32 793772030, 793772063) i32 @llvm.amdgcn.mbcnt.hi(i32 793772030, i32 793772030)
+; DEFAULT-NEXT: [[HI1:%.*]] = call range(i32 793772030, 793772063) i32 @llvm.amdgcn.mbcnt.hi(i32 [[MASK]], i32 793772030)
+; DEFAULT-NEXT: [[XOR:%.*]] = xor i32 [[HI0]], [[HI1]]
+; DEFAULT-NEXT: [[MIX:%.*]] = xor i32 [[XOR]], 1
+; DEFAULT-NEXT: ret i32 [[MIX]]
;
; WAVE64-LABEL: define i32 @mbcnt_hi_load_mask
; WAVE64-SAME: (ptr addrspace(1) [[IN:%.*]]) #[[ATTR1]] {
-; WAVE64-NEXT: ret i32 1
+; WAVE64-NEXT: [[MASK:%.*]] = load i32, ptr addrspace(1) [[IN]], align 4
+; WAVE64-NEXT: [[HI0:%.*]] = call range(i32 793772030, 793772063) i32 @llvm.amdgcn.mbcnt.hi(i32 793772030, i32 793772030)
+; WAVE64-NEXT: [[HI1:%.*]] = call range(i32 793772030, 793772063) i32 @llvm.amdgcn.mbcnt.hi(i32 [[MASK]], i32 793772030)
+; WAVE64-NEXT: [[XOR:%.*]] = xor i32 [[HI0]], [[HI1]]
+; WAVE64-NEXT: [[MIX:%.*]] = xor i32 [[XOR]], 1
+; WAVE64-NEXT: ret i32 [[MIX]]
;
%mask = load i32, ptr addrspace(1) %in, align 4
%hi0 = call i32 @llvm.amdgcn.mbcnt.hi(i32 793772030, i32 793772030)
>From 15aa5fe56d84429533c8766429809d4ba43b4d9e Mon Sep 17 00:00:00 2001
From: tmahatej_amdeng <teja.alaghari at amd.com>
Date: Thu, 1 Oct 2026 10:19:10 +0530
Subject: [PATCH 4/4] Revert peephole optimization for folding a zero mask to
base in mbcnt
---
.../Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp | 4 ----
llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll | 14 --------------
2 files changed, 18 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
index b29496a597684..4936b16148ce4 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstCombineIntrinsic.cpp
@@ -1744,10 +1744,6 @@ GCNTTIImpl::instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const {
return IC.replaceInstUsesWith(II, II.getArgOperand(1));
[[fallthrough]];
case Intrinsic::amdgcn_mbcnt_lo: {
- // No lanes contribute when the mask is zero.
- if (match(II.getArgOperand(0), m_Zero()))
- return IC.replaceInstUsesWith(II, II.getArgOperand(1));
-
ConstantRange AccRange =
computeConstantRange(II.getArgOperand(1),
/*ForSigned=*/false, IC.getSimplifyQuery());
diff --git a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
index 453f9c305de1a..d66f324031bd8 100644
--- a/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
+++ b/llvm/test/Transforms/InstCombine/AMDGPU/mbcnt.ll
@@ -292,20 +292,6 @@ define i32 @known_range_mbcnt_lo_refineable_range(i32 %unknown) {
ret i32 %lo
}
-; A zero mask contributes no lanes, so the call is the base.
-define i32 @mbcnt_hi_zero_mask() {
-; DEFAULT-LABEL: define i32 @mbcnt_hi_zero_mask() {
-; DEFAULT-NEXT: ret i32 47667
-;
-; WAVE64-LABEL: define i32 @mbcnt_hi_zero_mask
-; WAVE64-SAME: () #[[ATTR1]] {
-; WAVE64-NEXT: ret i32 47667
-;
- %hi = call i32 @llvm.amdgcn.mbcnt.hi(i32 0, i32 47666)
- %xor = xor i32 %hi, 1
- ret i32 %xor
-}
-
; A non-zero mask is lane-varying, so the call must not become a constant.
define i32 @mbcnt_lo_nonzero_mask() {
; DEFAULT-LABEL: define i32 @mbcnt_lo_nonzero_mask() {
More information about the llvm-commits
mailing list