[llvm] [AMDGPU] Fix umin(sffbh(x), bitwidth) fold when x may be all-ones (PR #201795)
Arseniy Obolenskiy via llvm-commits
llvm-commits at lists.llvm.org
Fri Jun 5 06:31:27 PDT 2026
https://github.com/aobolensk updated https://github.com/llvm/llvm-project/pull/201795
>From d16332ea3f070c33f822b458a78020249701ab6f Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Fri, 5 Jun 2026 11:15:31 +0200
Subject: [PATCH] [AMDGPU] Fix umin(sffbh(x), bitwidth) fold when x may be
all-ones
We can drop the umin clamp only if x is neither 0 nor -1. There is a problem with -1 check:
- old: The old guard !isAllOnes() is too weak because if the bit is not definitely unknown (could be either 0 or 1) this check will not work properly
- new: Known.Zero.getBoolValue() guarantees that there is at least one `0` bit which means that the value is definitely not -1 which is needed for this specific case
---
llvm/lib/Target/AMDGPU/SIISelLowering.cpp | 2 +-
llvm/test/CodeGen/AMDGPU/ctls.ll | 26 +++++++++++++++++++++++
2 files changed, 27 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index b393bb904e75b..d321b01a92d95 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -16280,7 +16280,7 @@ SDValue SITargetLowering::performMinMaxCombine(SDNode *N,
unsigned BitWidth = FfbhSrc.getValueType().getScalarSizeInBits();
if (Clamp >= BitWidth) {
KnownBits Known = DAG.computeKnownBits(FfbhSrc);
- if (Known.isNonZero() && !Known.isAllOnes())
+ if (Known.isNonZero() && Known.Zero.getBoolValue())
return Op0;
}
}
diff --git a/llvm/test/CodeGen/AMDGPU/ctls.ll b/llvm/test/CodeGen/AMDGPU/ctls.ll
index 00673b4a6d288..410b293d7f0e7 100644
--- a/llvm/test/CodeGen/AMDGPU/ctls.ll
+++ b/llvm/test/CodeGen/AMDGPU/ctls.ll
@@ -5,6 +5,7 @@
declare i32 @llvm.ctlz.i32(i32, i1)
declare i64 @llvm.ctlz.i64(i64, i1)
declare i32 @llvm.amdgcn.sffbh.i32(i32)
+declare i32 @llvm.umin.i32(i32, i32)
; Test that ctls(x) is lowered to umin(ffbh_i32(x), bitwidth) - 1
; ctls is formed by the DAG combiner from: ctlz(x ^ ashr(x, 31)) - 1
@@ -140,6 +141,31 @@ define i32 @ctls_i32_known_mixed_bits(i32 %x) {
ret i32 %d
}
+; Only bit 0 is known set; the high bits are unknown so x may be all-ones.
+; The umin clamp must be preserved (sffbh(-1) = -1 must clamp to bitwidth).
+define i32 @ctls_i32_maybe_all_ones(i32 %x) {
+; GFX6-LABEL: ctls_i32_maybe_all_ones:
+; GFX6: ; %bb.0:
+; GFX6-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX6-NEXT: v_or_b32_e32 v0, 1, v0
+; GFX6-NEXT: v_ffbh_i32_e32 v0, v0
+; GFX6-NEXT: v_min_u32_e32 v0, 32, v0
+; GFX6-NEXT: s_setpc_b64 s[30:31]
+;
+; GFX11-LABEL: ctls_i32_maybe_all_ones:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX11-NEXT: v_or_b32_e32 v0, 1, v0
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_cls_i32_e32 v0, v0
+; GFX11-NEXT: v_min_u32_e32 v0, 32, v0
+; GFX11-NEXT: s_setpc_b64 s[30:31]
+ %nz = or i32 %x, 1
+ %sffbh = call i32 @llvm.amdgcn.sffbh.i32(i32 %nz)
+ %r = call i32 @llvm.umin.i32(i32 %sffbh, i32 32)
+ ret i32 %r
+}
+
; test for i64 CTLS.
define i32 @ctls_i64(i64 %x) {
; GFX6-LABEL: ctls_i64:
More information about the llvm-commits
mailing list