[llvm] [NVPTX] Fix good bits calculation in bfe replacement (PR #225616)
Terence Ng via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 23 00:19:02 PDT 2026
https://github.com/TerenceNg03 created https://github.com/llvm/llvm-project/pull/225616
This PR changes the calculation of good bits from the bit width of the shift value to the bit width of the actual source value. This allows bfe replacement to be correctly triggered when the source size is larger than the shift value size. For example, when the source size is 64, the shift value size is 40, the good bits should be 24 instead of -8.
>From 3cb49a0020c6fd531918042d4bf11c81f189c038 Mon Sep 17 00:00:00 2001
From: TerenceNg03 <terenceng03 at icloud.com>
Date: Tue, 22 Sep 2026 08:37:02 +0000
Subject: [PATCH] [NVPTX] Fix shr/and width calculation in bfe replacement
---
llvm/lib/Target/NVPTX/NVPTXISelDAGToDAG.cpp | 2 +-
llvm/test/CodeGen/NVPTX/bfe.ll | 24 +++++++++++++++++++++
2 files changed, 25 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Target/NVPTX/NVPTXISelDAGToDAG.cpp b/llvm/lib/Target/NVPTX/NVPTXISelDAGToDAG.cpp
index 7411aa11824ff7..57bfb2c2c5bcc4 100644
--- a/llvm/lib/Target/NVPTX/NVPTXISelDAGToDAG.cpp
+++ b/llvm/lib/Target/NVPTX/NVPTXISelDAGToDAG.cpp
@@ -1820,7 +1820,7 @@ bool NVPTXDAGToDAGISel::tryBFE(SDNode *N) {
uint64_t StartVal = StartConst->getZExtValue();
// How many "good" bits do we have left? "good" is defined here as bits
// that exist in the original value, not shifted in.
- int64_t GoodBits = Start.getValueSizeInBits() - StartVal;
+ int64_t GoodBits = Val.getValueSizeInBits() - StartVal;
if (NumBits > GoodBits) {
// Do not handle the case where bits have been shifted in. In theory
// we could handle this, but the cost is likely higher than just
diff --git a/llvm/test/CodeGen/NVPTX/bfe.ll b/llvm/test/CodeGen/NVPTX/bfe.ll
index 644bf3606e8f64..79bf315cb4f0e1 100644
--- a/llvm/test/CodeGen/NVPTX/bfe.ll
+++ b/llvm/test/CodeGen/NVPTX/bfe.ll
@@ -99,6 +99,30 @@ define i64 @no_bfe_on_64bit_overflow(i64 %a) {
ret i64 %val1
}
+define i64 @bfe_i64_shift_amount_width(i64 %x) {
+; CHECK-O3-LABEL: bfe_i64_shift_amount_width(
+; CHECK-O3: {
+; CHECK-O3-NEXT: .reg .b64 %rd<2>;
+; CHECK-O3-EMPTY:
+; CHECK-O3-NEXT: // %bb.0:
+; CHECK-O3-NEXT: ld.param.b8 %rd1, [bfe_i64_shift_amount_width_param_0+5];
+; CHECK-O3-NEXT: st.param.b64 [func_retval0], %rd1;
+; CHECK-O3-NEXT: ret;
+;
+; CHECK-O0-LABEL: bfe_i64_shift_amount_width(
+; CHECK-O0: {
+; CHECK-O0-NEXT: .reg .b64 %rd<3>;
+; CHECK-O0-EMPTY:
+; CHECK-O0-NEXT: // %bb.0:
+; CHECK-O0-NEXT: ld.param.b64 %rd1, [bfe_i64_shift_amount_width_param_0];
+; CHECK-O0-NEXT: bfe.u64 %rd2, %rd1, 40, 8;
+; CHECK-O0-NEXT: st.param.b64 [func_retval0], %rd2;
+; CHECK-O0-NEXT: ret;
+ %shr = lshr i64 %x, 40
+ %and = and i64 %shr, 255
+ ret i64 %and
+}
+
define i64 @no_bfe_on_64bit_overflow_shr_and_pair(i64 %a) {
; CHECK-LABEL: no_bfe_on_64bit_overflow_shr_and_pair(
; CHECK: {
More information about the llvm-commits
mailing list