[llvm] [AMDGPU] Narrow an int to fp source that fits in half its width (PR #223107)
Dmitry Sidorov via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 13 10:22:31 PDT 2026
https://github.com/MrSidims updated https://github.com/llvm/llvm-project/pull/223107
>From bcd5f466379992855fa8188677cc1e147c883059 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Fri, 11 Sep 2026 03:41:06 +0200
Subject: [PATCH 1/3] [AMDGPU] Convert an i64 that fits in 32 bits from its low
half
An i64 whose high half is zero or only repeats the sign bit converts
like its low half, without the i64 expansion.
---
llvm/lib/Target/AMDGPU/AMDGPUISelLowering.cpp | 9 +
.../AMDGPU/fold-int-pow2-with-fmul-or-fdiv.ll | 28 +--
llvm/test/CodeGen/AMDGPU/sint_to_fp.f64.ll | 30 +--
llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll | 195 +++---------------
llvm/test/CodeGen/AMDGPU/uint_to_fp.f64.ll | 26 +--
llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll | 151 +++-----------
6 files changed, 83 insertions(+), 356 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUISelLowering.cpp b/llvm/lib/Target/AMDGPU/AMDGPUISelLowering.cpp
index 7fc2b116cca146..6a4b1048b53201 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUISelLowering.cpp
@@ -3674,6 +3674,15 @@ SDValue AMDGPUTargetLowering::lowerINT_TO_FPImpl(SDValue Op, SelectionDAG &DAG,
return DAG.getNode(CvtOpc, DL, DestVT, Ext);
}
+ // Narrow an i64 that fits in 32 bits.
+ if (SrcVT == MVT::i64 &&
+ (Signed ? DAG.ComputeNumSignBits(Src) > 32
+ : DAG.MaskedValueIsZero(Src, APInt::getHighBitsSet(64, 32)))) {
+ SDLoc DL(Op);
+ SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Src);
+ return DAG.getNode(CvtOpc, DL, DestVT, Trunc);
+ }
+
if (DestVT == MVT::bf16 || DestVT == MVT::f16)
return LowerINT_TO_FP16(Op, DAG, DestVT);
diff --git a/llvm/test/CodeGen/AMDGPU/fold-int-pow2-with-fmul-or-fdiv.ll b/llvm/test/CodeGen/AMDGPU/fold-int-pow2-with-fmul-or-fdiv.ll
index 0507e926ec6fe6..7165663bb2ad5f 100644
--- a/llvm/test/CodeGen/AMDGPU/fold-int-pow2-with-fmul-or-fdiv.ll
+++ b/llvm/test/CodeGen/AMDGPU/fold-int-pow2-with-fmul-or-fdiv.ll
@@ -558,16 +558,8 @@ define float @fmul_fly_pow_mul_min_pow2(i64 %cnt) nounwind {
; VI-NEXT: s_mov_b64 s[4:5], 0x2000
; VI-NEXT: v_cmp_gt_u64_e32 vcc, s[4:5], v[0:1]
; VI-NEXT: v_mov_b32_e32 v2, 0x2000
-; VI-NEXT: v_cndmask_b32_e32 v1, 0, v1, vcc
; VI-NEXT: v_cndmask_b32_e32 v0, v2, v0, vcc
-; VI-NEXT: v_ffbh_u32_e32 v2, v1
-; VI-NEXT: v_min_u32_e32 v2, 32, v2
-; VI-NEXT: v_lshlrev_b64 v[0:1], v2, v[0:1]
-; VI-NEXT: v_min_u32_e32 v0, 1, v0
-; VI-NEXT: v_or_b32_e32 v0, v1, v0
; VI-NEXT: v_cvt_f32_u32_e32 v0, v0
-; VI-NEXT: v_sub_u32_e32 v1, vcc, 32, v2
-; VI-NEXT: v_ldexp_f32 v0, v0, v1
; VI-NEXT: v_mul_f32_e32 v0, 0x41100000, v0
; VI-NEXT: s_setpc_b64 s[30:31]
;
@@ -576,16 +568,8 @@ define float @fmul_fly_pow_mul_min_pow2(i64 %cnt) nounwind {
; GFX10-NEXT: s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
; GFX10-NEXT: v_lshlrev_b64 v[0:1], v0, 8
; GFX10-NEXT: v_cmp_gt_u64_e32 vcc_lo, 0x2000, v[0:1]
-; GFX10-NEXT: v_cndmask_b32_e32 v1, 0, v1, vcc_lo
; GFX10-NEXT: v_cndmask_b32_e32 v0, 0x2000, v0, vcc_lo
-; GFX10-NEXT: v_ffbh_u32_e32 v2, v1
-; GFX10-NEXT: v_min_u32_e32 v2, 32, v2
-; GFX10-NEXT: v_lshlrev_b64 v[0:1], v2, v[0:1]
-; GFX10-NEXT: v_min_u32_e32 v0, 1, v0
-; GFX10-NEXT: v_or_b32_e32 v0, v1, v0
-; GFX10-NEXT: v_sub_nc_u32_e32 v1, 32, v2
; GFX10-NEXT: v_cvt_f32_u32_e32 v0, v0
-; GFX10-NEXT: v_ldexp_f32 v0, v0, v1
; GFX10-NEXT: v_mul_f32_e32 v0, 0x41100000, v0
; GFX10-NEXT: s_setpc_b64 s[30:31]
;
@@ -595,18 +579,8 @@ define float @fmul_fly_pow_mul_min_pow2(i64 %cnt) nounwind {
; GFX11-NEXT: v_lshlrev_b64 v[0:1], v0, 8
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(SKIP_1) | instid1(VALU_DEP_1)
; GFX11-NEXT: v_cmp_gt_u64_e32 vcc_lo, 0x2000, v[0:1]
-; GFX11-NEXT: v_dual_cndmask_b32 v1, 0, v1 :: v_dual_cndmask_b32 v0, 0x2000, v0
-; GFX11-NEXT: v_clz_i32_u32_e32 v2, v1
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-NEXT: v_min_u32_e32 v2, 32, v2
-; GFX11-NEXT: v_lshlrev_b64 v[0:1], v2, v[0:1]
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-NEXT: v_min_u32_e32 v0, 1, v0
-; GFX11-NEXT: v_or_b32_e32 v0, v1, v0
-; GFX11-NEXT: v_sub_nc_u32_e32 v1, 32, v2
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_2) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-NEXT: v_cndmask_b32_e32 v0, 0x2000, v0, vcc_lo
; GFX11-NEXT: v_cvt_f32_u32_e32 v0, v0
-; GFX11-NEXT: v_ldexp_f32 v0, v0, v1
; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-NEXT: v_mul_f32_e32 v0, 0x41100000, v0
; GFX11-NEXT: s_setpc_b64 s[30:31]
diff --git a/llvm/test/CodeGen/AMDGPU/sint_to_fp.f64.ll b/llvm/test/CodeGen/AMDGPU/sint_to_fp.f64.ll
index ca5f4ab792ecc8..6f5b0aec152c80 100644
--- a/llvm/test/CodeGen/AMDGPU/sint_to_fp.f64.ll
+++ b/llvm/test/CodeGen/AMDGPU/sint_to_fp.f64.ll
@@ -275,17 +275,12 @@ define amdgpu_kernel void @s_sint_to_fp_sext_i32_to_f64(ptr addrspace(1) %out, i
; CI-LABEL: s_sint_to_fp_sext_i32_to_f64:
; CI: ; %bb.0:
; CI-NEXT: s_load_dword s2, s[8:9], 0x2
+; CI-NEXT: s_load_dwordx2 s[0:1], s[8:9], 0x0
; CI-NEXT: s_add_i32 s12, s12, s17
; CI-NEXT: s_mov_b32 flat_scratch_lo, s13
; CI-NEXT: s_lshr_b32 flat_scratch_hi, s12, 8
; CI-NEXT: s_waitcnt lgkmcnt(0)
-; CI-NEXT: s_ashr_i32 s0, s2, 31
-; CI-NEXT: v_cvt_f64_i32_e32 v[0:1], s0
-; CI-NEXT: s_load_dwordx2 s[0:1], s[8:9], 0x0
-; CI-NEXT: v_cvt_f64_u32_e32 v[2:3], s2
-; CI-NEXT: v_ldexp_f64 v[0:1], v[0:1], 32
-; CI-NEXT: v_add_f64 v[0:1], v[0:1], v[2:3]
-; CI-NEXT: s_waitcnt lgkmcnt(0)
+; CI-NEXT: v_cvt_f64_i32_e32 v[0:1], s2
; CI-NEXT: v_mov_b32_e32 v3, s1
; CI-NEXT: v_mov_b32_e32 v2, s0
; CI-NEXT: flat_store_dwordx2 v[2:3], v[0:1]
@@ -293,18 +288,13 @@ define amdgpu_kernel void @s_sint_to_fp_sext_i32_to_f64(ptr addrspace(1) %out, i
;
; VI-LABEL: s_sint_to_fp_sext_i32_to_f64:
; VI: ; %bb.0:
-; VI-NEXT: s_load_dword s0, s[8:9], 0x8
+; VI-NEXT: s_load_dword s2, s[8:9], 0x8
+; VI-NEXT: s_load_dwordx2 s[0:1], s[8:9], 0x0
; VI-NEXT: s_add_i32 s12, s12, s17
; VI-NEXT: s_mov_b32 flat_scratch_lo, s13
; VI-NEXT: s_lshr_b32 flat_scratch_hi, s12, 8
; VI-NEXT: s_waitcnt lgkmcnt(0)
-; VI-NEXT: s_ashr_i32 s1, s0, 31
-; VI-NEXT: v_cvt_f64_i32_e32 v[0:1], s1
-; VI-NEXT: v_cvt_f64_u32_e32 v[2:3], s0
-; VI-NEXT: s_load_dwordx2 s[0:1], s[8:9], 0x0
-; VI-NEXT: v_ldexp_f64 v[0:1], v[0:1], 32
-; VI-NEXT: v_add_f64 v[0:1], v[0:1], v[2:3]
-; VI-NEXT: s_waitcnt lgkmcnt(0)
+; VI-NEXT: v_cvt_f64_i32_e32 v[0:1], s2
; VI-NEXT: v_mov_b32_e32 v3, s1
; VI-NEXT: v_mov_b32_e32 v2, s0
; VI-NEXT: flat_store_dwordx2 v[2:3], v[0:1]
@@ -314,14 +304,10 @@ define amdgpu_kernel void @s_sint_to_fp_sext_i32_to_f64(ptr addrspace(1) %out, i
; GFX942: ; %bb.0:
; GFX942-NEXT: s_load_dword s2, s[4:5], 0x8
; GFX942-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x0
-; GFX942-NEXT: v_mov_b32_e32 v4, 0
+; GFX942-NEXT: v_mov_b32_e32 v2, 0
; GFX942-NEXT: s_waitcnt lgkmcnt(0)
-; GFX942-NEXT: s_ashr_i32 s3, s2, 31
-; GFX942-NEXT: v_cvt_f64_i32_e32 v[0:1], s3
-; GFX942-NEXT: v_ldexp_f64 v[0:1], v[0:1], 32
-; GFX942-NEXT: v_cvt_f64_u32_e32 v[2:3], s2
-; GFX942-NEXT: v_add_f64 v[0:1], v[0:1], v[2:3]
-; GFX942-NEXT: global_store_dwordx2 v4, v[0:1], s[0:1]
+; GFX942-NEXT: v_cvt_f64_i32_e32 v[0:1], s2
+; GFX942-NEXT: global_store_dwordx2 v2, v[0:1], s[0:1]
; GFX942-NEXT: s_endpgm
%wide = sext i32 %in to i64
%result = sitofp i64 %wide to double
diff --git a/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll b/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll
index c32ba64166091e..857cda5324c084 100644
--- a/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll
@@ -1199,24 +1199,12 @@ define amdgpu_kernel void @s_sint_to_fp_sext_i32_to_f16(ptr addrspace(1) %out, i
; GFX6-LABEL: s_sint_to_fp_sext_i32_to_f16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_load_dword s0, s[4:5], 0xb
+; GFX6-NEXT: s_mov_b32 s3, 0xf000
+; GFX6-NEXT: s_mov_b32 s2, -1
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_ashr_i32 s1, s0, 31
-; GFX6-NEXT: s_xor_b32 s2, s0, s1
-; GFX6-NEXT: s_flbit_i32 s3, s1
-; GFX6-NEXT: s_ashr_i32 s2, s2, 31
-; GFX6-NEXT: s_add_i32 s2, s2, 32
-; GFX6-NEXT: s_add_i32 s3, s3, -1
-; GFX6-NEXT: s_min_u32 s2, s3, s2
-; GFX6-NEXT: s_lshl_b64 s[0:1], s[0:1], s2
-; GFX6-NEXT: s_min_u32 s0, s0, 1
-; GFX6-NEXT: s_or_b32 s0, s1, s0
; GFX6-NEXT: v_cvt_f32_i32_e32 v0, s0
-; GFX6-NEXT: s_sub_i32 s2, 32, s2
; GFX6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
-; GFX6-NEXT: s_mov_b32 s3, 0xf000
-; GFX6-NEXT: v_ldexp_f32_e64 v0, v0, s2
; GFX6-NEXT: v_cvt_f16_f32_e32 v0, v0
-; GFX6-NEXT: s_mov_b32 s2, -1
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
; GFX6-NEXT: buffer_store_short v0, off, s[0:3], 0
; GFX6-NEXT: s_endpgm
@@ -1225,20 +1213,8 @@ define amdgpu_kernel void @s_sint_to_fp_sext_i32_to_f16(ptr addrspace(1) %out, i
; GFX8: ; %bb.0:
; GFX8-NEXT: s_load_dword s0, s[4:5], 0x2c
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_ashr_i32 s1, s0, 31
-; GFX8-NEXT: s_xor_b32 s2, s0, s1
-; GFX8-NEXT: s_flbit_i32 s3, s1
-; GFX8-NEXT: s_ashr_i32 s2, s2, 31
-; GFX8-NEXT: s_add_i32 s2, s2, 32
-; GFX8-NEXT: s_add_i32 s3, s3, -1
-; GFX8-NEXT: s_min_u32 s2, s3, s2
-; GFX8-NEXT: s_lshl_b64 s[0:1], s[0:1], s2
-; GFX8-NEXT: s_min_u32 s0, s0, 1
-; GFX8-NEXT: s_or_b32 s0, s1, s0
; GFX8-NEXT: v_cvt_f32_i32_e32 v0, s0
; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
-; GFX8-NEXT: s_sub_i32 s2, 32, s2
-; GFX8-NEXT: v_ldexp_f32 v0, v0, s2
; GFX8-NEXT: v_cvt_f16_f32_e32 v2, v0
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
; GFX8-NEXT: v_mov_b32_e32 v0, s0
@@ -1248,60 +1224,28 @@ define amdgpu_kernel void @s_sint_to_fp_sext_i32_to_f16(ptr addrspace(1) %out, i
;
; GFX11-TRUE16-LABEL: s_sint_to_fp_sext_i32_to_f16:
; GFX11-TRUE16: ; %bb.0:
-; GFX11-TRUE16-NEXT: s_load_b32 s0, s[4:5], 0x2c
+; GFX11-TRUE16-NEXT: s_clause 0x1
+; GFX11-TRUE16-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, 0
; GFX11-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-TRUE16-NEXT: s_ashr_i32 s1, s0, 31
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
-; GFX11-TRUE16-NEXT: s_xor_b32 s2, s0, s1
-; GFX11-TRUE16-NEXT: s_cls_i32 s3, s1
-; GFX11-TRUE16-NEXT: s_ashr_i32 s2, s2, 31
-; GFX11-TRUE16-NEXT: s_add_i32 s3, s3, -1
-; GFX11-TRUE16-NEXT: s_add_i32 s2, s2, 32
-; GFX11-TRUE16-NEXT: s_min_u32 s6, s3, s2
-; GFX11-TRUE16-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
-; GFX11-TRUE16-NEXT: s_lshl_b64 s[0:1], s[0:1], s6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-TRUE16-NEXT: s_min_u32 s0, s0, 1
-; GFX11-TRUE16-NEXT: s_or_b32 s0, s1, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11-TRUE16-NEXT: v_cvt_f32_i32_e32 v0, s0
-; GFX11-TRUE16-NEXT: s_sub_i32 s0, 32, s6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
-; GFX11-TRUE16-NEXT: v_ldexp_f32 v0, v0, s0
+; GFX11-TRUE16-NEXT: v_cvt_f32_i32_e32 v0, s2
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
-; GFX11-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-TRUE16-NEXT: global_store_b16 v1, v0, s[2:3]
+; GFX11-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX11-TRUE16-NEXT: s_endpgm
;
; GFX11-FAKE16-LABEL: s_sint_to_fp_sext_i32_to_f16:
; GFX11-FAKE16: ; %bb.0:
-; GFX11-FAKE16-NEXT: s_load_b32 s0, s[4:5], 0x2c
+; GFX11-FAKE16-NEXT: s_clause 0x1
+; GFX11-FAKE16-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, 0
; GFX11-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-FAKE16-NEXT: s_ashr_i32 s1, s0, 31
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
-; GFX11-FAKE16-NEXT: s_xor_b32 s2, s0, s1
-; GFX11-FAKE16-NEXT: s_cls_i32 s3, s1
-; GFX11-FAKE16-NEXT: s_ashr_i32 s2, s2, 31
-; GFX11-FAKE16-NEXT: s_add_i32 s3, s3, -1
-; GFX11-FAKE16-NEXT: s_add_i32 s2, s2, 32
-; GFX11-FAKE16-NEXT: s_min_u32 s6, s3, s2
-; GFX11-FAKE16-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
-; GFX11-FAKE16-NEXT: s_lshl_b64 s[0:1], s[0:1], s6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-FAKE16-NEXT: s_min_u32 s0, s0, 1
-; GFX11-FAKE16-NEXT: s_or_b32 s0, s1, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11-FAKE16-NEXT: v_cvt_f32_i32_e32 v0, s0
-; GFX11-FAKE16-NEXT: s_sub_i32 s0, 32, s6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
-; GFX11-FAKE16-NEXT: v_ldexp_f32 v0, v0, s0
+; GFX11-FAKE16-NEXT: v_cvt_f32_i32_e32 v0, s2
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
-; GFX11-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-FAKE16-NEXT: global_store_b16 v1, v0, s[2:3]
+; GFX11-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX11-FAKE16-NEXT: s_endpgm
%wide = sext i32 %in to i64
%result = sitofp i64 %wide to half
@@ -1314,73 +1258,33 @@ define amdgpu_kernel void @s_sint_to_fp_sext_i32_to_f32(ptr addrspace(1) %out, i
; GFX6: ; %bb.0:
; GFX6-NEXT: s_load_dword s2, s[4:5], 0xb
; GFX6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; GFX6-NEXT: s_mov_b32 s3, 0xf000
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_ashr_i32 s3, s2, 31
-; GFX6-NEXT: s_xor_b32 s4, s2, s3
-; GFX6-NEXT: s_flbit_i32 s5, s3
-; GFX6-NEXT: s_ashr_i32 s4, s4, 31
-; GFX6-NEXT: s_add_i32 s4, s4, 32
-; GFX6-NEXT: s_add_i32 s5, s5, -1
-; GFX6-NEXT: s_min_u32 s4, s5, s4
-; GFX6-NEXT: s_lshl_b64 s[2:3], s[2:3], s4
-; GFX6-NEXT: s_min_u32 s2, s2, 1
-; GFX6-NEXT: s_or_b32 s2, s3, s2
; GFX6-NEXT: v_cvt_f32_i32_e32 v0, s2
-; GFX6-NEXT: s_sub_i32 s4, 32, s4
-; GFX6-NEXT: s_mov_b32 s3, 0xf000
; GFX6-NEXT: s_mov_b32 s2, -1
-; GFX6-NEXT: v_ldexp_f32_e64 v0, v0, s4
; GFX6-NEXT: buffer_store_dword v0, off, s[0:3], 0
; GFX6-NEXT: s_endpgm
;
; GFX8-LABEL: s_sint_to_fp_sext_i32_to_f32:
; GFX8: ; %bb.0:
-; GFX8-NEXT: s_load_dword s0, s[4:5], 0x2c
-; GFX8-NEXT: s_load_dwordx2 s[2:3], s[4:5], 0x24
+; GFX8-NEXT: s_load_dword s2, s[4:5], 0x2c
+; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_ashr_i32 s1, s0, 31
-; GFX8-NEXT: s_xor_b32 s4, s0, s1
-; GFX8-NEXT: s_flbit_i32 s5, s1
-; GFX8-NEXT: s_ashr_i32 s4, s4, 31
-; GFX8-NEXT: s_add_i32 s5, s5, -1
-; GFX8-NEXT: s_add_i32 s4, s4, 32
-; GFX8-NEXT: s_min_u32 s4, s5, s4
-; GFX8-NEXT: s_lshl_b64 s[0:1], s[0:1], s4
-; GFX8-NEXT: s_min_u32 s0, s0, 1
-; GFX8-NEXT: s_or_b32 s0, s1, s0
-; GFX8-NEXT: v_cvt_f32_i32_e32 v0, s0
-; GFX8-NEXT: s_sub_i32 s0, 32, s4
-; GFX8-NEXT: v_mov_b32_e32 v1, s3
-; GFX8-NEXT: v_ldexp_f32 v2, v0, s0
-; GFX8-NEXT: v_mov_b32_e32 v0, s2
+; GFX8-NEXT: v_cvt_f32_i32_e32 v2, s2
+; GFX8-NEXT: v_mov_b32_e32 v0, s0
+; GFX8-NEXT: v_mov_b32_e32 v1, s1
; GFX8-NEXT: flat_store_dword v[0:1], v2
; GFX8-NEXT: s_endpgm
;
; GFX11-LABEL: s_sint_to_fp_sext_i32_to_f32:
; GFX11: ; %bb.0:
; GFX11-NEXT: s_clause 0x1
-; GFX11-NEXT: s_load_b32 s0, s[4:5], 0x2c
-; GFX11-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
-; GFX11-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v0, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-NEXT: s_ashr_i32 s1, s0, 31
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(SKIP_4) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_xor_b32 s6, s0, s1
-; GFX11-NEXT: s_cls_i32 s5, s1
-; GFX11-NEXT: s_ashr_i32 s4, s6, 31
-; GFX11-NEXT: s_add_i32 s5, s5, -1
-; GFX11-NEXT: s_add_i32 s4, s4, 32
-; GFX11-NEXT: s_min_u32 s4, s5, s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_lshl_b64 s[0:1], s[0:1], s4
-; GFX11-NEXT: s_min_u32 s0, s0, 1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_or_b32 s0, s1, s0
-; GFX11-NEXT: v_cvt_f32_i32_e32 v0, s0
-; GFX11-NEXT: s_sub_i32 s0, 32, s4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: v_ldexp_f32 v0, v0, s0
-; GFX11-NEXT: global_store_b32 v1, v0, s[2:3]
+; GFX11-NEXT: v_cvt_f32_i32_e32 v1, s2
+; GFX11-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-NEXT: s_endpgm
%wide = sext i32 %in to i64
%result = sitofp i64 %wide to float
@@ -1395,22 +1299,9 @@ define amdgpu_kernel void @s_sint_to_fp_33_sign_bits_to_f32(ptr addrspace(1) %ou
; GFX6-NEXT: s_mov_b32 s7, 0xf000
; GFX6-NEXT: s_mov_b32 s6, -1
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_ashr_i32 s5, s3, 31
-; GFX6-NEXT: s_mov_b32 s4, s3
-; GFX6-NEXT: s_xor_b32 s3, s3, s5
-; GFX6-NEXT: s_flbit_i32 s2, s5
-; GFX6-NEXT: s_ashr_i32 s3, s3, 31
-; GFX6-NEXT: s_add_i32 s2, s2, -1
-; GFX6-NEXT: s_add_i32 s3, s3, 32
-; GFX6-NEXT: s_min_u32 s8, s2, s3
-; GFX6-NEXT: s_lshl_b64 s[2:3], s[4:5], s8
-; GFX6-NEXT: s_min_u32 s2, s2, 1
-; GFX6-NEXT: s_or_b32 s2, s3, s2
-; GFX6-NEXT: v_cvt_f32_i32_e32 v0, s2
+; GFX6-NEXT: v_cvt_f32_i32_e32 v0, s3
; GFX6-NEXT: s_mov_b32 s4, s0
-; GFX6-NEXT: s_sub_i32 s0, 32, s8
; GFX6-NEXT: s_mov_b32 s5, s1
-; GFX6-NEXT: v_ldexp_f32_e64 v0, v0, s0
; GFX6-NEXT: buffer_store_dword v0, off, s[4:7], 0
; GFX6-NEXT: s_endpgm
;
@@ -1418,49 +1309,19 @@ define amdgpu_kernel void @s_sint_to_fp_33_sign_bits_to_f32(ptr addrspace(1) %ou
; GFX8: ; %bb.0:
; GFX8-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_ashr_i32 s5, s3, 31
-; GFX8-NEXT: s_mov_b32 s4, s3
-; GFX8-NEXT: s_xor_b32 s3, s3, s5
-; GFX8-NEXT: s_flbit_i32 s2, s5
-; GFX8-NEXT: s_ashr_i32 s3, s3, 31
-; GFX8-NEXT: s_add_i32 s2, s2, -1
-; GFX8-NEXT: s_add_i32 s3, s3, 32
-; GFX8-NEXT: s_min_u32 s6, s2, s3
-; GFX8-NEXT: s_lshl_b64 s[2:3], s[4:5], s6
-; GFX8-NEXT: s_min_u32 s2, s2, 1
-; GFX8-NEXT: s_or_b32 s2, s3, s2
-; GFX8-NEXT: v_cvt_f32_i32_e32 v2, s2
+; GFX8-NEXT: v_cvt_f32_i32_e32 v2, s3
; GFX8-NEXT: v_mov_b32_e32 v0, s0
-; GFX8-NEXT: s_sub_i32 s0, 32, s6
; GFX8-NEXT: v_mov_b32_e32 v1, s1
-; GFX8-NEXT: v_ldexp_f32 v2, v2, s0
; GFX8-NEXT: flat_store_dword v[0:1], v2
; GFX8-NEXT: s_endpgm
;
; GFX11-LABEL: s_sint_to_fp_33_sign_bits_to_f32:
; GFX11: ; %bb.0:
; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX11-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-NEXT: v_mov_b32_e32 v0, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-NEXT: s_ashr_i32 s5, s3, 31
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11-NEXT: s_xor_b32 s2, s3, s5
-; GFX11-NEXT: s_cls_i32 s4, s5
-; GFX11-NEXT: s_ashr_i32 s2, s2, 31
-; GFX11-NEXT: s_add_i32 s6, s4, -1
-; GFX11-NEXT: s_add_i32 s2, s2, 32
-; GFX11-NEXT: s_mov_b32 s4, s3
-; GFX11-NEXT: s_min_u32 s6, s6, s2
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_lshl_b64 s[2:3], s[4:5], s6
-; GFX11-NEXT: s_min_u32 s2, s2, 1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_or_b32 s2, s3, s2
-; GFX11-NEXT: v_cvt_f32_i32_e32 v0, s2
-; GFX11-NEXT: s_sub_i32 s2, 32, s6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: v_ldexp_f32 v0, v0, s2
-; GFX11-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-NEXT: v_cvt_f32_i32_e32 v1, s3
+; GFX11-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-NEXT: s_endpgm
%narrow = ashr i64 %in, 32
%result = sitofp i64 %narrow to float
diff --git a/llvm/test/CodeGen/AMDGPU/uint_to_fp.f64.ll b/llvm/test/CodeGen/AMDGPU/uint_to_fp.f64.ll
index 4cd063725ed48c..2ee17e10c79d57 100644
--- a/llvm/test/CodeGen/AMDGPU/uint_to_fp.f64.ll
+++ b/llvm/test/CodeGen/AMDGPU/uint_to_fp.f64.ll
@@ -292,14 +292,11 @@ define amdgpu_kernel void @s_uint_to_fp_zext_i32_to_f64(ptr addrspace(1) %out, i
; SI: ; %bb.0:
; SI-NEXT: s_load_dword s2, s[8:9], 0x2
; SI-NEXT: s_load_dwordx2 s[0:1], s[8:9], 0x0
-; SI-NEXT: v_cvt_f64_u32_e32 v[0:1], 0
; SI-NEXT: s_add_i32 s12, s12, s17
; SI-NEXT: s_mov_b32 flat_scratch_lo, s13
-; SI-NEXT: s_waitcnt lgkmcnt(0)
-; SI-NEXT: v_cvt_f64_u32_e32 v[2:3], s2
-; SI-NEXT: v_ldexp_f64 v[0:1], v[0:1], 32
; SI-NEXT: s_lshr_b32 flat_scratch_hi, s12, 8
-; SI-NEXT: v_add_f64 v[0:1], v[0:1], v[2:3]
+; SI-NEXT: s_waitcnt lgkmcnt(0)
+; SI-NEXT: v_cvt_f64_u32_e32 v[0:1], s2
; SI-NEXT: v_mov_b32_e32 v3, s1
; SI-NEXT: v_mov_b32_e32 v2, s0
; SI-NEXT: flat_store_dwordx2 v[2:3], v[0:1]
@@ -307,17 +304,13 @@ define amdgpu_kernel void @s_uint_to_fp_zext_i32_to_f64(ptr addrspace(1) %out, i
;
; VI-LABEL: s_uint_to_fp_zext_i32_to_f64:
; VI: ; %bb.0:
-; VI-NEXT: v_cvt_f64_u32_e32 v[0:1], 0
-; VI-NEXT: s_load_dword s0, s[8:9], 0x8
+; VI-NEXT: s_load_dword s2, s[8:9], 0x8
+; VI-NEXT: s_load_dwordx2 s[0:1], s[8:9], 0x0
; VI-NEXT: s_add_i32 s12, s12, s17
; VI-NEXT: s_mov_b32 flat_scratch_lo, s13
-; VI-NEXT: v_ldexp_f64 v[0:1], v[0:1], 32
; VI-NEXT: s_lshr_b32 flat_scratch_hi, s12, 8
; VI-NEXT: s_waitcnt lgkmcnt(0)
-; VI-NEXT: v_cvt_f64_u32_e32 v[2:3], s0
-; VI-NEXT: s_load_dwordx2 s[0:1], s[8:9], 0x0
-; VI-NEXT: v_add_f64 v[0:1], v[0:1], v[2:3]
-; VI-NEXT: s_waitcnt lgkmcnt(0)
+; VI-NEXT: v_cvt_f64_u32_e32 v[0:1], s2
; VI-NEXT: v_mov_b32_e32 v3, s1
; VI-NEXT: v_mov_b32_e32 v2, s0
; VI-NEXT: flat_store_dwordx2 v[2:3], v[0:1]
@@ -327,13 +320,10 @@ define amdgpu_kernel void @s_uint_to_fp_zext_i32_to_f64(ptr addrspace(1) %out, i
; GFX942: ; %bb.0:
; GFX942-NEXT: s_load_dword s2, s[4:5], 0x8
; GFX942-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x0
-; GFX942-NEXT: v_cvt_f64_u32_e32 v[0:1], 0
-; GFX942-NEXT: v_ldexp_f64 v[0:1], v[0:1], 32
-; GFX942-NEXT: v_mov_b32_e32 v4, 0
+; GFX942-NEXT: v_mov_b32_e32 v2, 0
; GFX942-NEXT: s_waitcnt lgkmcnt(0)
-; GFX942-NEXT: v_cvt_f64_u32_e32 v[2:3], s2
-; GFX942-NEXT: v_add_f64 v[0:1], v[0:1], v[2:3]
-; GFX942-NEXT: global_store_dwordx2 v4, v[0:1], s[0:1]
+; GFX942-NEXT: v_cvt_f64_u32_e32 v[0:1], s2
+; GFX942-NEXT: global_store_dwordx2 v2, v[0:1], s[0:1]
; GFX942-NEXT: s_endpgm
%wide = zext i32 %in to i64
%result = uitofp i64 %wide to double
diff --git a/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll b/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll
index b1e0ec060fbd88..e5cee26af0ed2b 100644
--- a/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll
@@ -966,20 +966,12 @@ define amdgpu_kernel void @s_uint_to_fp_zext_i32_to_f16(ptr addrspace(1) %out, i
; GFX6-LABEL: s_uint_to_fp_zext_i32_to_f16:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_load_dword s0, s[4:5], 0xb
-; GFX6-NEXT: s_flbit_i32_b32 s2, 0
-; GFX6-NEXT: s_mov_b32 s1, 0
-; GFX6-NEXT: s_min_u32 s2, s2, 32
; GFX6-NEXT: s_mov_b32 s3, 0xf000
+; GFX6-NEXT: s_mov_b32 s2, -1
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_lshl_b64 s[0:1], s[0:1], s2
-; GFX6-NEXT: s_min_u32 s0, s0, 1
-; GFX6-NEXT: s_or_b32 s0, s1, s0
; GFX6-NEXT: v_cvt_f32_u32_e32 v0, s0
-; GFX6-NEXT: s_sub_i32 s2, 32, s2
; GFX6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
-; GFX6-NEXT: v_ldexp_f32_e64 v0, v0, s2
; GFX6-NEXT: v_cvt_f16_f32_e32 v0, v0
-; GFX6-NEXT: s_mov_b32 s2, -1
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
; GFX6-NEXT: buffer_store_short v0, off, s[0:3], 0
; GFX6-NEXT: s_endpgm
@@ -987,17 +979,9 @@ define amdgpu_kernel void @s_uint_to_fp_zext_i32_to_f16(ptr addrspace(1) %out, i
; GFX8-LABEL: s_uint_to_fp_zext_i32_to_f16:
; GFX8: ; %bb.0:
; GFX8-NEXT: s_load_dword s0, s[4:5], 0x2c
-; GFX8-NEXT: s_flbit_i32_b32 s2, 0
-; GFX8-NEXT: s_mov_b32 s1, 0
-; GFX8-NEXT: s_min_u32 s2, s2, 32
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_lshl_b64 s[0:1], s[0:1], s2
-; GFX8-NEXT: s_min_u32 s0, s0, 1
-; GFX8-NEXT: s_or_b32 s0, s1, s0
; GFX8-NEXT: v_cvt_f32_u32_e32 v0, s0
; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
-; GFX8-NEXT: s_sub_i32 s2, 32, s2
-; GFX8-NEXT: v_ldexp_f32 v0, v0, s2
; GFX8-NEXT: v_cvt_f16_f32_e32 v2, v0
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
; GFX8-NEXT: v_mov_b32_e32 v0, s0
@@ -1007,48 +991,28 @@ define amdgpu_kernel void @s_uint_to_fp_zext_i32_to_f16(ptr addrspace(1) %out, i
;
; GFX11-TRUE16-LABEL: s_uint_to_fp_zext_i32_to_f16:
; GFX11-TRUE16: ; %bb.0:
-; GFX11-TRUE16-NEXT: s_load_b32 s0, s[4:5], 0x2c
-; GFX11-TRUE16-NEXT: s_clz_i32_u32 s2, 0
-; GFX11-TRUE16-NEXT: s_mov_b32 s1, 0
-; GFX11-TRUE16-NEXT: s_min_u32 s6, s2, 32
-; GFX11-TRUE16-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
+; GFX11-TRUE16-NEXT: s_clause 0x1
+; GFX11-TRUE16-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-TRUE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-TRUE16-NEXT: v_mov_b32_e32 v1, 0
; GFX11-TRUE16-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-TRUE16-NEXT: s_lshl_b64 s[0:1], s[0:1], s6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-TRUE16-NEXT: s_min_u32 s0, s0, 1
-; GFX11-TRUE16-NEXT: s_or_b32 s0, s1, s0
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11-TRUE16-NEXT: v_cvt_f32_u32_e32 v0, s0
-; GFX11-TRUE16-NEXT: s_sub_i32 s0, 32, s6
-; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
-; GFX11-TRUE16-NEXT: v_ldexp_f32 v0, v0, s0
+; GFX11-TRUE16-NEXT: v_cvt_f32_u32_e32 v0, s2
; GFX11-TRUE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-TRUE16-NEXT: v_cvt_f16_f32_e32 v0.l, v0
-; GFX11-TRUE16-NEXT: global_store_b16 v1, v0, s[2:3]
+; GFX11-TRUE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX11-TRUE16-NEXT: s_endpgm
;
; GFX11-FAKE16-LABEL: s_uint_to_fp_zext_i32_to_f16:
; GFX11-FAKE16: ; %bb.0:
-; GFX11-FAKE16-NEXT: s_load_b32 s0, s[4:5], 0x2c
-; GFX11-FAKE16-NEXT: s_clz_i32_u32 s2, 0
-; GFX11-FAKE16-NEXT: s_mov_b32 s1, 0
-; GFX11-FAKE16-NEXT: s_min_u32 s6, s2, 32
-; GFX11-FAKE16-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
+; GFX11-FAKE16-NEXT: s_clause 0x1
+; GFX11-FAKE16-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-FAKE16-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-FAKE16-NEXT: v_mov_b32_e32 v1, 0
; GFX11-FAKE16-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-FAKE16-NEXT: s_lshl_b64 s[0:1], s[0:1], s6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-FAKE16-NEXT: s_min_u32 s0, s0, 1
-; GFX11-FAKE16-NEXT: s_or_b32 s0, s1, s0
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11-FAKE16-NEXT: v_cvt_f32_u32_e32 v0, s0
-; GFX11-FAKE16-NEXT: s_sub_i32 s0, 32, s6
-; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
-; GFX11-FAKE16-NEXT: v_ldexp_f32 v0, v0, s0
+; GFX11-FAKE16-NEXT: v_cvt_f32_u32_e32 v0, s2
; GFX11-FAKE16-NEXT: s_delay_alu instid0(VALU_DEP_1)
; GFX11-FAKE16-NEXT: v_cvt_f16_f32_e32 v0, v0
-; GFX11-FAKE16-NEXT: global_store_b16 v1, v0, s[2:3]
+; GFX11-FAKE16-NEXT: global_store_b16 v1, v0, s[0:1]
; GFX11-FAKE16-NEXT: s_endpgm
%wide = zext i32 %in to i64
%result = uitofp i64 %wide to half
@@ -1061,60 +1025,33 @@ define amdgpu_kernel void @s_uint_to_fp_zext_i32_to_f32(ptr addrspace(1) %out, i
; GFX6: ; %bb.0:
; GFX6-NEXT: s_load_dword s2, s[4:5], 0xb
; GFX6-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
-; GFX6-NEXT: s_flbit_i32_b32 s4, 0
-; GFX6-NEXT: s_mov_b32 s3, 0
-; GFX6-NEXT: s_min_u32 s4, s4, 32
+; GFX6-NEXT: s_mov_b32 s3, 0xf000
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_lshl_b64 s[2:3], s[2:3], s4
-; GFX6-NEXT: s_min_u32 s2, s2, 1
-; GFX6-NEXT: s_or_b32 s2, s3, s2
; GFX6-NEXT: v_cvt_f32_u32_e32 v0, s2
-; GFX6-NEXT: s_sub_i32 s4, 32, s4
-; GFX6-NEXT: s_mov_b32 s3, 0xf000
; GFX6-NEXT: s_mov_b32 s2, -1
-; GFX6-NEXT: v_ldexp_f32_e64 v0, v0, s4
; GFX6-NEXT: buffer_store_dword v0, off, s[0:3], 0
; GFX6-NEXT: s_endpgm
;
; GFX8-LABEL: s_uint_to_fp_zext_i32_to_f32:
; GFX8: ; %bb.0:
-; GFX8-NEXT: s_load_dword s0, s[4:5], 0x2c
-; GFX8-NEXT: s_load_dwordx2 s[2:3], s[4:5], 0x24
-; GFX8-NEXT: s_flbit_i32_b32 s4, 0
-; GFX8-NEXT: s_mov_b32 s1, 0
-; GFX8-NEXT: s_min_u32 s4, s4, 32
+; GFX8-NEXT: s_load_dword s2, s[4:5], 0x2c
+; GFX8-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_lshl_b64 s[0:1], s[0:1], s4
-; GFX8-NEXT: s_min_u32 s0, s0, 1
-; GFX8-NEXT: s_or_b32 s0, s1, s0
-; GFX8-NEXT: v_cvt_f32_u32_e32 v0, s0
-; GFX8-NEXT: s_sub_i32 s0, 32, s4
-; GFX8-NEXT: v_mov_b32_e32 v1, s3
-; GFX8-NEXT: v_ldexp_f32 v2, v0, s0
-; GFX8-NEXT: v_mov_b32_e32 v0, s2
+; GFX8-NEXT: v_cvt_f32_u32_e32 v2, s2
+; GFX8-NEXT: v_mov_b32_e32 v0, s0
+; GFX8-NEXT: v_mov_b32_e32 v1, s1
; GFX8-NEXT: flat_store_dword v[0:1], v2
; GFX8-NEXT: s_endpgm
;
; GFX11-LABEL: s_uint_to_fp_zext_i32_to_f32:
; GFX11: ; %bb.0:
; GFX11-NEXT: s_clause 0x1
-; GFX11-NEXT: s_load_b32 s0, s[4:5], 0x2c
-; GFX11-NEXT: s_load_b64 s[2:3], s[4:5], 0x24
-; GFX11-NEXT: s_clz_i32_u32 s4, 0
-; GFX11-NEXT: s_mov_b32 s1, 0
-; GFX11-NEXT: s_min_u32 s4, s4, 32
-; GFX11-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v0, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-NEXT: s_lshl_b64 s[0:1], s[0:1], s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_min_u32 s0, s0, 1
-; GFX11-NEXT: s_or_b32 s0, s1, s0
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11-NEXT: v_cvt_f32_u32_e32 v0, s0
-; GFX11-NEXT: s_sub_i32 s0, 32, s4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: v_ldexp_f32 v0, v0, s0
-; GFX11-NEXT: global_store_b32 v1, v0, s[2:3]
+; GFX11-NEXT: v_cvt_f32_u32_e32 v1, s2
+; GFX11-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-NEXT: s_endpgm
%wide = zext i32 %in to i64
%result = uitofp i64 %wide to float
@@ -1126,21 +1063,12 @@ define amdgpu_kernel void @s_uint_to_fp_32_zero_bits_to_f32(ptr addrspace(1) %ou
; GFX6-LABEL: s_uint_to_fp_32_zero_bits_to_f32:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x9
-; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_flbit_i32_b32 s2, 0
-; GFX6-NEXT: s_mov_b32 s5, 0
-; GFX6-NEXT: s_min_u32 s8, s2, 32
; GFX6-NEXT: s_mov_b32 s7, 0xf000
-; GFX6-NEXT: s_mov_b32 s4, s3
-; GFX6-NEXT: s_lshl_b64 s[2:3], s[4:5], s8
-; GFX6-NEXT: s_min_u32 s2, s2, 1
-; GFX6-NEXT: s_or_b32 s2, s3, s2
-; GFX6-NEXT: v_cvt_f32_u32_e32 v0, s2
-; GFX6-NEXT: s_mov_b32 s4, s0
-; GFX6-NEXT: s_sub_i32 s0, 32, s8
; GFX6-NEXT: s_mov_b32 s6, -1
+; GFX6-NEXT: s_waitcnt lgkmcnt(0)
+; GFX6-NEXT: v_cvt_f32_u32_e32 v0, s3
+; GFX6-NEXT: s_mov_b32 s4, s0
; GFX6-NEXT: s_mov_b32 s5, s1
-; GFX6-NEXT: v_ldexp_f32_e64 v0, v0, s0
; GFX6-NEXT: buffer_store_dword v0, off, s[4:7], 0
; GFX6-NEXT: s_endpgm
;
@@ -1148,40 +1076,19 @@ define amdgpu_kernel void @s_uint_to_fp_32_zero_bits_to_f32(ptr addrspace(1) %ou
; GFX8: ; %bb.0:
; GFX8-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_flbit_i32_b32 s2, 0
-; GFX8-NEXT: s_mov_b32 s5, 0
-; GFX8-NEXT: s_min_u32 s6, s2, 32
-; GFX8-NEXT: s_mov_b32 s4, s3
-; GFX8-NEXT: s_lshl_b64 s[2:3], s[4:5], s6
-; GFX8-NEXT: s_min_u32 s2, s2, 1
-; GFX8-NEXT: s_or_b32 s2, s3, s2
-; GFX8-NEXT: v_cvt_f32_u32_e32 v2, s2
+; GFX8-NEXT: v_cvt_f32_u32_e32 v2, s3
; GFX8-NEXT: v_mov_b32_e32 v0, s0
-; GFX8-NEXT: s_sub_i32 s0, 32, s6
; GFX8-NEXT: v_mov_b32_e32 v1, s1
-; GFX8-NEXT: v_ldexp_f32 v2, v2, s0
; GFX8-NEXT: flat_store_dword v[0:1], v2
; GFX8-NEXT: s_endpgm
;
; GFX11-LABEL: s_uint_to_fp_32_zero_bits_to_f32:
; GFX11: ; %bb.0:
; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v0, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-NEXT: s_clz_i32_u32 s2, 0
-; GFX11-NEXT: s_mov_b32 s5, 0
-; GFX11-NEXT: s_min_u32 s6, s2, 32
-; GFX11-NEXT: v_mov_b32_e32 v1, 0
-; GFX11-NEXT: s_mov_b32 s4, s3
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_lshl_b64 s[2:3], s[4:5], s6
-; GFX11-NEXT: s_min_u32 s2, s2, 1
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_or_b32 s2, s3, s2
-; GFX11-NEXT: v_cvt_f32_u32_e32 v0, s2
-; GFX11-NEXT: s_sub_i32 s2, 32, s6
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: v_ldexp_f32 v0, v0, s2
-; GFX11-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-NEXT: v_cvt_f32_u32_e32 v1, s3
+; GFX11-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-NEXT: s_endpgm
%narrow = lshr i64 %in, 32
%result = uitofp i64 %narrow to float
>From 5e955fb502ad9818d1532ba4dc4a02194a8c9d6d Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 13 Sep 2026 17:59:56 +0200
Subject: [PATCH 2/3] [NFC][AMDGPU] Add int to fp tests for trunc nsw, nuw and
nneg sources
---
llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll | 74 ++++++++++++++++
llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll | 99 ++++++++++++++++++++++
2 files changed, 173 insertions(+)
diff --git a/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll b/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll
index 857cda5324c084..7317da7e3c329f 100644
--- a/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll
@@ -1406,6 +1406,80 @@ define amdgpu_kernel void @s_sint_to_fp_32_sign_bits_to_f32(ptr addrspace(1) %ou
ret void
}
+define amdgpu_kernel void @s_sint_to_fp_trunc_nsw_i64_to_f32(ptr addrspace(1) %out, i64 %in) #0 {
+; GFX6-LABEL: s_sint_to_fp_trunc_nsw_i64_to_f32:
+; GFX6: ; %bb.0:
+; GFX6-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x9
+; GFX6-NEXT: s_mov_b32 s7, 0xf000
+; GFX6-NEXT: s_mov_b32 s6, -1
+; GFX6-NEXT: s_waitcnt lgkmcnt(0)
+; GFX6-NEXT: s_xor_b32 s5, s2, s3
+; GFX6-NEXT: s_flbit_i32 s4, s3
+; GFX6-NEXT: s_ashr_i32 s5, s5, 31
+; GFX6-NEXT: s_add_i32 s4, s4, -1
+; GFX6-NEXT: s_add_i32 s5, s5, 32
+; GFX6-NEXT: s_min_u32 s8, s4, s5
+; GFX6-NEXT: s_lshl_b64 s[2:3], s[2:3], s8
+; GFX6-NEXT: s_min_u32 s2, s2, 1
+; GFX6-NEXT: s_or_b32 s2, s3, s2
+; GFX6-NEXT: v_cvt_f32_i32_e32 v0, s2
+; GFX6-NEXT: s_mov_b32 s4, s0
+; GFX6-NEXT: s_sub_i32 s0, 32, s8
+; GFX6-NEXT: s_mov_b32 s5, s1
+; GFX6-NEXT: v_ldexp_f32_e64 v0, v0, s0
+; GFX6-NEXT: buffer_store_dword v0, off, s[4:7], 0
+; GFX6-NEXT: s_endpgm
+;
+; GFX8-LABEL: s_sint_to_fp_trunc_nsw_i64_to_f32:
+; GFX8: ; %bb.0:
+; GFX8-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX8-NEXT: s_waitcnt lgkmcnt(0)
+; GFX8-NEXT: s_xor_b32 s5, s2, s3
+; GFX8-NEXT: s_flbit_i32 s4, s3
+; GFX8-NEXT: s_ashr_i32 s5, s5, 31
+; GFX8-NEXT: s_add_i32 s4, s4, -1
+; GFX8-NEXT: s_add_i32 s5, s5, 32
+; GFX8-NEXT: s_min_u32 s4, s4, s5
+; GFX8-NEXT: s_lshl_b64 s[2:3], s[2:3], s4
+; GFX8-NEXT: s_min_u32 s2, s2, 1
+; GFX8-NEXT: s_or_b32 s2, s3, s2
+; GFX8-NEXT: v_cvt_f32_i32_e32 v2, s2
+; GFX8-NEXT: v_mov_b32_e32 v0, s0
+; GFX8-NEXT: s_sub_i32 s0, 32, s4
+; GFX8-NEXT: v_mov_b32_e32 v1, s1
+; GFX8-NEXT: v_ldexp_f32 v2, v2, s0
+; GFX8-NEXT: flat_store_dword v[0:1], v2
+; GFX8-NEXT: s_endpgm
+;
+; GFX11-LABEL: s_sint_to_fp_trunc_nsw_i64_to_f32:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: s_xor_b32 s4, s2, s3
+; GFX11-NEXT: s_cls_i32 s5, s3
+; GFX11-NEXT: s_ashr_i32 s4, s4, 31
+; GFX11-NEXT: s_add_i32 s5, s5, -1
+; GFX11-NEXT: s_add_i32 s4, s4, 32
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_min_u32 s4, s5, s4
+; GFX11-NEXT: s_lshl_b64 s[2:3], s[2:3], s4
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_min_u32 s2, s2, 1
+; GFX11-NEXT: s_or_b32 s2, s3, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: v_cvt_f32_i32_e32 v0, s2
+; GFX11-NEXT: s_sub_i32 s2, 32, s4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: v_ldexp_f32 v0, v0, s2
+; GFX11-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-NEXT: s_endpgm
+ %narrow = trunc nsw i64 %in to i32
+ %result = sitofp i32 %narrow to float
+ store float %result, ptr addrspace(1) %out
+ ret void
+}
+
declare i32 @llvm.amdgcn.workitem.id.x() #1
attributes #0 = { nounwind }
diff --git a/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll b/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll
index e5cee26af0ed2b..85055f51699949 100644
--- a/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll
@@ -1161,6 +1161,105 @@ define amdgpu_kernel void @s_uint_to_fp_31_zero_bits_to_f32(ptr addrspace(1) %ou
ret void
}
+define amdgpu_kernel void @s_uint_to_fp_nneg_32_zero_bits_to_f32(ptr addrspace(1) %out, i64 %in) #0 {
+; GFX6-LABEL: s_uint_to_fp_nneg_32_zero_bits_to_f32:
+; GFX6: ; %bb.0:
+; GFX6-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x9
+; GFX6-NEXT: s_mov_b32 s7, 0xf000
+; GFX6-NEXT: s_mov_b32 s6, -1
+; GFX6-NEXT: s_waitcnt lgkmcnt(0)
+; GFX6-NEXT: v_cvt_f32_u32_e32 v0, s3
+; GFX6-NEXT: s_mov_b32 s4, s0
+; GFX6-NEXT: s_mov_b32 s5, s1
+; GFX6-NEXT: buffer_store_dword v0, off, s[4:7], 0
+; GFX6-NEXT: s_endpgm
+;
+; GFX8-LABEL: s_uint_to_fp_nneg_32_zero_bits_to_f32:
+; GFX8: ; %bb.0:
+; GFX8-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX8-NEXT: s_waitcnt lgkmcnt(0)
+; GFX8-NEXT: v_cvt_f32_u32_e32 v2, s3
+; GFX8-NEXT: v_mov_b32_e32 v0, s0
+; GFX8-NEXT: v_mov_b32_e32 v1, s1
+; GFX8-NEXT: flat_store_dword v[0:1], v2
+; GFX8-NEXT: s_endpgm
+;
+; GFX11-LABEL: s_uint_to_fp_nneg_32_zero_bits_to_f32:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v0, 0
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: v_cvt_f32_u32_e32 v1, s3
+; GFX11-NEXT: global_store_b32 v0, v1, s[0:1]
+; GFX11-NEXT: s_endpgm
+ %narrow = lshr i64 %in, 32
+ %result = uitofp nneg i64 %narrow to float
+ store float %result, ptr addrspace(1) %out
+ ret void
+}
+
+define amdgpu_kernel void @s_uint_to_fp_trunc_nuw_i64_to_f32(ptr addrspace(1) %out, i64 %in) #0 {
+; GFX6-LABEL: s_uint_to_fp_trunc_nuw_i64_to_f32:
+; GFX6: ; %bb.0:
+; GFX6-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x9
+; GFX6-NEXT: s_mov_b32 s7, 0xf000
+; GFX6-NEXT: s_mov_b32 s6, -1
+; GFX6-NEXT: s_waitcnt lgkmcnt(0)
+; GFX6-NEXT: s_flbit_i32_b32 s4, s3
+; GFX6-NEXT: s_min_u32 s8, s4, 32
+; GFX6-NEXT: s_lshl_b64 s[2:3], s[2:3], s8
+; GFX6-NEXT: s_min_u32 s2, s2, 1
+; GFX6-NEXT: s_or_b32 s2, s3, s2
+; GFX6-NEXT: v_cvt_f32_u32_e32 v0, s2
+; GFX6-NEXT: s_mov_b32 s4, s0
+; GFX6-NEXT: s_sub_i32 s0, 32, s8
+; GFX6-NEXT: s_mov_b32 s5, s1
+; GFX6-NEXT: v_ldexp_f32_e64 v0, v0, s0
+; GFX6-NEXT: buffer_store_dword v0, off, s[4:7], 0
+; GFX6-NEXT: s_endpgm
+;
+; GFX8-LABEL: s_uint_to_fp_trunc_nuw_i64_to_f32:
+; GFX8: ; %bb.0:
+; GFX8-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GFX8-NEXT: s_waitcnt lgkmcnt(0)
+; GFX8-NEXT: s_flbit_i32_b32 s4, s3
+; GFX8-NEXT: s_min_u32 s4, s4, 32
+; GFX8-NEXT: s_lshl_b64 s[2:3], s[2:3], s4
+; GFX8-NEXT: s_min_u32 s2, s2, 1
+; GFX8-NEXT: s_or_b32 s2, s3, s2
+; GFX8-NEXT: v_cvt_f32_u32_e32 v2, s2
+; GFX8-NEXT: v_mov_b32_e32 v0, s0
+; GFX8-NEXT: s_sub_i32 s0, 32, s4
+; GFX8-NEXT: v_mov_b32_e32 v1, s1
+; GFX8-NEXT: v_ldexp_f32 v2, v2, s0
+; GFX8-NEXT: flat_store_dword v[0:1], v2
+; GFX8-NEXT: s_endpgm
+;
+; GFX11-LABEL: s_uint_to_fp_trunc_nuw_i64_to_f32:
+; GFX11: ; %bb.0:
+; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
+; GFX11-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-NEXT: s_clz_i32_u32 s4, s3
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_min_u32 s4, s4, 32
+; GFX11-NEXT: s_lshl_b64 s[2:3], s[2:3], s4
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: s_min_u32 s2, s2, 1
+; GFX11-NEXT: s_or_b32 s2, s3, s2
+; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
+; GFX11-NEXT: v_cvt_f32_u32_e32 v0, s2
+; GFX11-NEXT: s_sub_i32 s2, 32, s4
+; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
+; GFX11-NEXT: v_ldexp_f32 v0, v0, s2
+; GFX11-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-NEXT: s_endpgm
+ %narrow = trunc nuw i64 %in to i32
+ %result = uitofp i32 %narrow to float
+ store float %result, ptr addrspace(1) %out
+ ret void
+}
+
declare i32 @llvm.amdgcn.workitem.id.x() #1
attributes #0 = { nounwind }
>From 7612849e8f6124503f1f24961ec441ee3644481f Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 13 Sep 2026 17:59:56 +0200
Subject: [PATCH 3/3] move to dag combiner
---
llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp | 33 ++++++++++++
llvm/lib/Target/AMDGPU/AMDGPUISelLowering.cpp | 9 ----
llvm/lib/Target/AMDGPU/SIISelLowering.cpp | 3 ++
llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll | 51 +++----------------
llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll | 39 +++-----------
5 files changed, 48 insertions(+), 87 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
index ca292fc81afa68..1ceb54c22c1f34 100644
--- a/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/DAGCombiner.cpp
@@ -20542,6 +20542,33 @@ static SDValue foldFPToIntToFP(SDNode *N, const SDLoc &DL, SelectionDAG &DAG,
return Result;
}
+// Narrow an integer source that fits in half its width when the target finds
+// the wide type undesirable for the conversion.
+static SDValue narrowIntToFPSource(SDNode *N, const SDLoc &DL,
+ SelectionDAG &DAG,
+ const TargetLowering &TLI) {
+ unsigned Opc = N->getOpcode();
+ SDValue N0 = N->getOperand(0);
+ EVT OpVT = N0.getValueType();
+ if (!OpVT.isScalarInteger() || !TLI.isTypeLegal(OpVT) ||
+ TLI.isTypeDesirableForOp(Opc, OpVT))
+ return SDValue();
+
+ EVT HalfVT = OpVT.getHalfSizedIntegerVT(*DAG.getContext());
+ if (!TLI.isTypeDesirableForOp(Opc, HalfVT) ||
+ !TLI.isOperationLegalOrCustom(Opc, HalfVT))
+ return SDValue();
+
+ unsigned SrcBits = Opc == ISD::SINT_TO_FP
+ ? DAG.ComputeMaxSignificantBits(N0)
+ : DAG.computeKnownBits(N0).countMaxActiveBits();
+ if (SrcBits > HalfVT.getSizeInBits())
+ return SDValue();
+
+ return DAG.getNode(Opc, DL, N->getValueType(0),
+ DAG.getNode(ISD::TRUNCATE, DL, HalfVT, N0));
+}
+
SDValue DAGCombiner::visitSINT_TO_FP(SDNode *N) {
SDValue N0 = N->getOperand(0);
EVT VT = N->getValueType(0);
@@ -20593,6 +20620,9 @@ SDValue DAGCombiner::visitSINT_TO_FP(SDNode *N) {
N0.getOperand(0).getValueType()))
return DAG.getNode(ISD::SINT_TO_FP, DL, VT, N0.getOperand(0));
+ if (SDValue Narrow = narrowIntToFPSource(N, DL, DAG, TLI))
+ return Narrow;
+
return SDValue();
}
@@ -20636,6 +20666,9 @@ SDValue DAGCombiner::visitUINT_TO_FP(SDNode *N) {
N0.getOperand(0).getValueType()))
return DAG.getNode(ISD::UINT_TO_FP, DL, VT, N0.getOperand(0));
+ if (SDValue Narrow = narrowIntToFPSource(N, DL, DAG, TLI))
+ return Narrow;
+
return SDValue();
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUISelLowering.cpp b/llvm/lib/Target/AMDGPU/AMDGPUISelLowering.cpp
index 6a4b1048b53201..7fc2b116cca146 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUISelLowering.cpp
@@ -3674,15 +3674,6 @@ SDValue AMDGPUTargetLowering::lowerINT_TO_FPImpl(SDValue Op, SelectionDAG &DAG,
return DAG.getNode(CvtOpc, DL, DestVT, Ext);
}
- // Narrow an i64 that fits in 32 bits.
- if (SrcVT == MVT::i64 &&
- (Signed ? DAG.ComputeNumSignBits(Src) > 32
- : DAG.MaskedValueIsZero(Src, APInt::getHighBitsSet(64, 32)))) {
- SDLoc DL(Op);
- SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Src);
- return DAG.getNode(CvtOpc, DL, DestVT, Trunc);
- }
-
if (DestVT == MVT::bf16 || DestVT == MVT::f16)
return LowerINT_TO_FP16(Op, DAG, DestVT);
diff --git a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
index 5a26953096c4bd..8c9715f35a4bb7 100644
--- a/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIISelLowering.cpp
@@ -2485,6 +2485,9 @@ bool SITargetLowering::isTypeDesirableForOp(unsigned Op, EVT VT) const {
if (VT == MVT::i1 && Op == ISD::SETCC)
return false;
+ if (VT == MVT::i64 && (Op == ISD::SINT_TO_FP || Op == ISD::UINT_TO_FP))
+ return false;
+
return TargetLowering::isTypeDesirableForOp(Op, VT);
}
diff --git a/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll b/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll
index 7317da7e3c329f..8f9db6db1e7eed 100644
--- a/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/sint_to_fp.i64.ll
@@ -1410,69 +1410,30 @@ define amdgpu_kernel void @s_sint_to_fp_trunc_nsw_i64_to_f32(ptr addrspace(1) %o
; GFX6-LABEL: s_sint_to_fp_trunc_nsw_i64_to_f32:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x9
-; GFX6-NEXT: s_mov_b32 s7, 0xf000
-; GFX6-NEXT: s_mov_b32 s6, -1
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_xor_b32 s5, s2, s3
-; GFX6-NEXT: s_flbit_i32 s4, s3
-; GFX6-NEXT: s_ashr_i32 s5, s5, 31
-; GFX6-NEXT: s_add_i32 s4, s4, -1
-; GFX6-NEXT: s_add_i32 s5, s5, 32
-; GFX6-NEXT: s_min_u32 s8, s4, s5
-; GFX6-NEXT: s_lshl_b64 s[2:3], s[2:3], s8
-; GFX6-NEXT: s_min_u32 s2, s2, 1
-; GFX6-NEXT: s_or_b32 s2, s3, s2
+; GFX6-NEXT: s_mov_b32 s3, 0xf000
; GFX6-NEXT: v_cvt_f32_i32_e32 v0, s2
-; GFX6-NEXT: s_mov_b32 s4, s0
-; GFX6-NEXT: s_sub_i32 s0, 32, s8
-; GFX6-NEXT: s_mov_b32 s5, s1
-; GFX6-NEXT: v_ldexp_f32_e64 v0, v0, s0
-; GFX6-NEXT: buffer_store_dword v0, off, s[4:7], 0
+; GFX6-NEXT: s_mov_b32 s2, -1
+; GFX6-NEXT: buffer_store_dword v0, off, s[0:3], 0
; GFX6-NEXT: s_endpgm
;
; GFX8-LABEL: s_sint_to_fp_trunc_nsw_i64_to_f32:
; GFX8: ; %bb.0:
; GFX8-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_xor_b32 s5, s2, s3
-; GFX8-NEXT: s_flbit_i32 s4, s3
-; GFX8-NEXT: s_ashr_i32 s5, s5, 31
-; GFX8-NEXT: s_add_i32 s4, s4, -1
-; GFX8-NEXT: s_add_i32 s5, s5, 32
-; GFX8-NEXT: s_min_u32 s4, s4, s5
-; GFX8-NEXT: s_lshl_b64 s[2:3], s[2:3], s4
-; GFX8-NEXT: s_min_u32 s2, s2, 1
-; GFX8-NEXT: s_or_b32 s2, s3, s2
; GFX8-NEXT: v_cvt_f32_i32_e32 v2, s2
; GFX8-NEXT: v_mov_b32_e32 v0, s0
-; GFX8-NEXT: s_sub_i32 s0, 32, s4
; GFX8-NEXT: v_mov_b32_e32 v1, s1
-; GFX8-NEXT: v_ldexp_f32 v2, v2, s0
; GFX8-NEXT: flat_store_dword v[0:1], v2
; GFX8-NEXT: s_endpgm
;
; GFX11-LABEL: s_sint_to_fp_trunc_nsw_i64_to_f32:
; GFX11: ; %bb.0:
; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX11-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-NEXT: v_mov_b32_e32 v0, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-NEXT: s_xor_b32 s4, s2, s3
-; GFX11-NEXT: s_cls_i32 s5, s3
-; GFX11-NEXT: s_ashr_i32 s4, s4, 31
-; GFX11-NEXT: s_add_i32 s5, s5, -1
-; GFX11-NEXT: s_add_i32 s4, s4, 32
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_min_u32 s4, s5, s4
-; GFX11-NEXT: s_lshl_b64 s[2:3], s[2:3], s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_min_u32 s2, s2, 1
-; GFX11-NEXT: s_or_b32 s2, s3, s2
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11-NEXT: v_cvt_f32_i32_e32 v0, s2
-; GFX11-NEXT: s_sub_i32 s2, 32, s4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: v_ldexp_f32 v0, v0, s2
-; GFX11-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-NEXT: v_cvt_f32_i32_e32 v1, s2
+; GFX11-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-NEXT: s_endpgm
%narrow = trunc nsw i64 %in to i32
%result = sitofp i32 %narrow to float
diff --git a/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll b/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll
index 85055f51699949..69bd48af1284a0 100644
--- a/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll
+++ b/llvm/test/CodeGen/AMDGPU/uint_to_fp.i64.ll
@@ -1202,57 +1202,30 @@ define amdgpu_kernel void @s_uint_to_fp_trunc_nuw_i64_to_f32(ptr addrspace(1) %o
; GFX6-LABEL: s_uint_to_fp_trunc_nuw_i64_to_f32:
; GFX6: ; %bb.0:
; GFX6-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x9
-; GFX6-NEXT: s_mov_b32 s7, 0xf000
-; GFX6-NEXT: s_mov_b32 s6, -1
; GFX6-NEXT: s_waitcnt lgkmcnt(0)
-; GFX6-NEXT: s_flbit_i32_b32 s4, s3
-; GFX6-NEXT: s_min_u32 s8, s4, 32
-; GFX6-NEXT: s_lshl_b64 s[2:3], s[2:3], s8
-; GFX6-NEXT: s_min_u32 s2, s2, 1
-; GFX6-NEXT: s_or_b32 s2, s3, s2
+; GFX6-NEXT: s_mov_b32 s3, 0xf000
; GFX6-NEXT: v_cvt_f32_u32_e32 v0, s2
-; GFX6-NEXT: s_mov_b32 s4, s0
-; GFX6-NEXT: s_sub_i32 s0, 32, s8
-; GFX6-NEXT: s_mov_b32 s5, s1
-; GFX6-NEXT: v_ldexp_f32_e64 v0, v0, s0
-; GFX6-NEXT: buffer_store_dword v0, off, s[4:7], 0
+; GFX6-NEXT: s_mov_b32 s2, -1
+; GFX6-NEXT: buffer_store_dword v0, off, s[0:3], 0
; GFX6-NEXT: s_endpgm
;
; GFX8-LABEL: s_uint_to_fp_trunc_nuw_i64_to_f32:
; GFX8: ; %bb.0:
; GFX8-NEXT: s_load_dwordx4 s[0:3], s[4:5], 0x24
; GFX8-NEXT: s_waitcnt lgkmcnt(0)
-; GFX8-NEXT: s_flbit_i32_b32 s4, s3
-; GFX8-NEXT: s_min_u32 s4, s4, 32
-; GFX8-NEXT: s_lshl_b64 s[2:3], s[2:3], s4
-; GFX8-NEXT: s_min_u32 s2, s2, 1
-; GFX8-NEXT: s_or_b32 s2, s3, s2
; GFX8-NEXT: v_cvt_f32_u32_e32 v2, s2
; GFX8-NEXT: v_mov_b32_e32 v0, s0
-; GFX8-NEXT: s_sub_i32 s0, 32, s4
; GFX8-NEXT: v_mov_b32_e32 v1, s1
-; GFX8-NEXT: v_ldexp_f32 v2, v2, s0
; GFX8-NEXT: flat_store_dword v[0:1], v2
; GFX8-NEXT: s_endpgm
;
; GFX11-LABEL: s_uint_to_fp_trunc_nuw_i64_to_f32:
; GFX11: ; %bb.0:
; GFX11-NEXT: s_load_b128 s[0:3], s[4:5], 0x24
-; GFX11-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-NEXT: v_mov_b32_e32 v0, 0
; GFX11-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-NEXT: s_clz_i32_u32 s4, s3
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_min_u32 s4, s4, 32
-; GFX11-NEXT: s_lshl_b64 s[2:3], s[2:3], s4
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1) | instskip(NEXT) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: s_min_u32 s2, s2, 1
-; GFX11-NEXT: s_or_b32 s2, s3, s2
-; GFX11-NEXT: s_delay_alu instid0(SALU_CYCLE_1)
-; GFX11-NEXT: v_cvt_f32_u32_e32 v0, s2
-; GFX11-NEXT: s_sub_i32 s2, 32, s4
-; GFX11-NEXT: s_delay_alu instid0(VALU_DEP_1) | instid1(SALU_CYCLE_1)
-; GFX11-NEXT: v_ldexp_f32 v0, v0, s2
-; GFX11-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-NEXT: v_cvt_f32_u32_e32 v1, s2
+; GFX11-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-NEXT: s_endpgm
%narrow = trunc nuw i64 %in to i32
%result = uitofp i32 %narrow to float
More information about the llvm-commits
mailing list