[llvm] [AMDGPU] Update patterns for v_cvt_flr and v_cvt_rpi (PR #177962)
Mirko BrkuĊĦanin via llvm-commits
llvm-commits at lists.llvm.org
Mon Jan 26 07:14:15 PST 2026
https://github.com/mbrkusanin updated https://github.com/llvm/llvm-project/pull/177962
>From eec6ea9cc15781dba4618a19c50ac4710c3acca2 Mon Sep 17 00:00:00 2001
From: Mirko Brkusanin <Mirko.Brkusanin at amd.com>
Date: Mon, 26 Jan 2026 15:59:57 +0100
Subject: [PATCH 1/2] Update tests to use nnan flag and not
-enable-no-nans-fp-math
---
llvm/test/CodeGen/AMDGPU/cvt_flr_i32_f32.ll | 353 +++++++++++++++++---
llvm/test/CodeGen/AMDGPU/cvt_rpi_i32_f32.ll | 318 ++++++++++++++++--
2 files changed, 596 insertions(+), 75 deletions(-)
diff --git a/llvm/test/CodeGen/AMDGPU/cvt_flr_i32_f32.ll b/llvm/test/CodeGen/AMDGPU/cvt_flr_i32_f32.ll
index 0974ce99aee36..dcba86b04a934 100644
--- a/llvm/test/CodeGen/AMDGPU/cvt_flr_i32_f32.ll
+++ b/llvm/test/CodeGen/AMDGPU/cvt_flr_i32_f32.ll
@@ -1,82 +1,357 @@
-; RUN: llc -mtriple=amdgcn < %s | FileCheck -check-prefix=SI-SAFE -check-prefix=SI -check-prefix=FUNC %s
-; RUN: llc -mtriple=amdgcn -enable-no-nans-fp-math < %s | FileCheck -check-prefix=SI-NONAN -check-prefix=SI -check-prefix=FUNC %s
-; RUN: llc -mtriple=amdgcn -mcpu=tonga < %s | FileCheck -check-prefix=SI -check-prefix=FUNC %s
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgcn < %s | FileCheck -check-prefix=SI-SDAG %s
+; RUN: llc -mtriple=amdgcn -global-isel < %s | FileCheck -check-prefix=SI-GISEL %s
+; RUN: llc -mtriple=amdgcn -mcpu=gfx1100 < %s | FileCheck -check-prefix=GFX11-SDAG %s
+; RUN: llc -mtriple=amdgcn -mcpu=gfx1100 -global-isel < %s | FileCheck -check-prefix=GFX11-GISEL %s
declare float @llvm.fabs.f32(float) #1
declare float @llvm.floor.f32(float) #1
-; FUNC-LABEL: {{^}}cvt_flr_i32_f32_0:
-; SI-SAFE-NOT: v_cvt_flr_i32_f32
-; SI-NOT: add
-; SI-NONAN: v_cvt_flr_i32_f32_e32 v{{[0-9]+}}, s{{[0-9]+}}
-; SI: s_endpgm
define amdgpu_kernel void @cvt_flr_i32_f32_0(ptr addrspace(1) %out, float %x) #0 {
- %floor = call float @llvm.floor.f32(float %x) #1
+; SI-SDAG-LABEL: cvt_flr_i32_f32_0:
+; SI-SDAG: ; %bb.0:
+; SI-SDAG-NEXT: s_load_dword s6, s[4:5], 0xb
+; SI-SDAG-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
+; SI-SDAG-NEXT: s_mov_b32 s2, -1
+; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; SI-SDAG-NEXT: v_floor_f32_e32 v0, s6
+; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-SDAG-NEXT: s_endpgm
+;
+; SI-GISEL-LABEL: cvt_flr_i32_f32_0:
+; SI-GISEL: ; %bb.0:
+; SI-GISEL-NEXT: s_load_dword s3, s[4:5], 0xb
+; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-GISEL-NEXT: s_mov_b32 s2, -1
+; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; SI-GISEL-NEXT: v_floor_f32_e32 v0, s3
+; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
+; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-GISEL-NEXT: s_endpgm
+;
+; GFX11-SDAG-LABEL: cvt_flr_i32_f32_0:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_clause 0x1
+; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, s2
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: s_endpgm
+;
+; GFX11-GISEL-LABEL: cvt_flr_i32_f32_0:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_clause 0x1
+; GFX11-GISEL-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, s2
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-GISEL-NEXT: s_endpgm
+ %floor = call nnan float @llvm.floor.f32(float %x) #1
%cvt = fptosi float %floor to i32
store i32 %cvt, ptr addrspace(1) %out
ret void
}
-; FUNC-LABEL: {{^}}cvt_flr_i32_f32_1:
-; SI: v_add_f32_e64 [[TMP:v[0-9]+]], s{{[0-9]+}}, 1.0
-; SI-SAFE-NOT: v_cvt_flr_i32_f32
-; SI-NONAN: v_cvt_flr_i32_f32_e32 v{{[0-9]+}}, [[TMP]]
-; SI: s_endpgm
define amdgpu_kernel void @cvt_flr_i32_f32_1(ptr addrspace(1) %out, float %x) #0 {
+; SI-SDAG-LABEL: cvt_flr_i32_f32_1:
+; SI-SDAG: ; %bb.0:
+; SI-SDAG-NEXT: s_load_dword s6, s[4:5], 0xb
+; SI-SDAG-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
+; SI-SDAG-NEXT: s_mov_b32 s2, -1
+; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; SI-SDAG-NEXT: v_add_f32_e64 v0, s6, 1.0
+; SI-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-SDAG-NEXT: s_endpgm
+;
+; SI-GISEL-LABEL: cvt_flr_i32_f32_1:
+; SI-GISEL: ; %bb.0:
+; SI-GISEL-NEXT: s_load_dword s3, s[4:5], 0xb
+; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-GISEL-NEXT: s_mov_b32 s2, -1
+; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; SI-GISEL-NEXT: v_add_f32_e64 v0, s3, 1.0
+; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
+; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-GISEL-NEXT: s_endpgm
+;
+; GFX11-SDAG-LABEL: cvt_flr_i32_f32_1:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_clause 0x1
+; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-SDAG-NEXT: v_add_f32_e64 v0, s2, 1.0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: s_endpgm
+;
+; GFX11-GISEL-LABEL: cvt_flr_i32_f32_1:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_clause 0x1
+; GFX11-GISEL-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_add_f32_e64 v0, s2, 1.0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-GISEL-NEXT: s_endpgm
%fadd = fadd float %x, 1.0
- %floor = call float @llvm.floor.f32(float %fadd) #1
+ %floor = call nnan float @llvm.floor.f32(float %fadd) #1
%cvt = fptosi float %floor to i32
store i32 %cvt, ptr addrspace(1) %out
ret void
}
-; FUNC-LABEL: {{^}}cvt_flr_i32_f32_fabs:
-; SI-NOT: add
-; SI-SAFE-NOT: v_cvt_flr_i32_f32
-; SI-NONAN: v_cvt_flr_i32_f32_e64 v{{[0-9]+}}, |s{{[0-9]+}}|
-; SI: s_endpgm
define amdgpu_kernel void @cvt_flr_i32_f32_fabs(ptr addrspace(1) %out, float %x) #0 {
+; SI-SDAG-LABEL: cvt_flr_i32_f32_fabs:
+; SI-SDAG: ; %bb.0:
+; SI-SDAG-NEXT: s_load_dword s6, s[4:5], 0xb
+; SI-SDAG-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
+; SI-SDAG-NEXT: s_mov_b32 s2, -1
+; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; SI-SDAG-NEXT: v_floor_f32_e64 v0, |s6|
+; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-SDAG-NEXT: s_endpgm
+;
+; SI-GISEL-LABEL: cvt_flr_i32_f32_fabs:
+; SI-GISEL: ; %bb.0:
+; SI-GISEL-NEXT: s_load_dword s3, s[4:5], 0xb
+; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-GISEL-NEXT: s_mov_b32 s2, -1
+; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; SI-GISEL-NEXT: v_floor_f32_e64 v0, |s3|
+; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
+; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-GISEL-NEXT: s_endpgm
+;
+; GFX11-SDAG-LABEL: cvt_flr_i32_f32_fabs:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_clause 0x1
+; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-SDAG-NEXT: v_floor_f32_e64 v0, |s2|
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: s_endpgm
+;
+; GFX11-GISEL-LABEL: cvt_flr_i32_f32_fabs:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_clause 0x1
+; GFX11-GISEL-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_floor_f32_e64 v0, |s2|
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-GISEL-NEXT: s_endpgm
%x.fabs = call float @llvm.fabs.f32(float %x) #1
- %floor = call float @llvm.floor.f32(float %x.fabs) #1
+ %floor = call nnan float @llvm.floor.f32(float %x.fabs) #1
%cvt = fptosi float %floor to i32
store i32 %cvt, ptr addrspace(1) %out
ret void
}
-; FUNC-LABEL: {{^}}cvt_flr_i32_f32_fneg:
-; SI-NOT: add
-; SI-SAFE-NOT: v_cvt_flr_i32_f32
-; SI-NONAN: v_cvt_flr_i32_f32_e64 v{{[0-9]+}}, -s{{[0-9]+}}
-; SI: s_endpgm
define amdgpu_kernel void @cvt_flr_i32_f32_fneg(ptr addrspace(1) %out, float %x) #0 {
+; SI-SDAG-LABEL: cvt_flr_i32_f32_fneg:
+; SI-SDAG: ; %bb.0:
+; SI-SDAG-NEXT: s_load_dword s6, s[4:5], 0xb
+; SI-SDAG-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
+; SI-SDAG-NEXT: s_mov_b32 s2, -1
+; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; SI-SDAG-NEXT: v_floor_f32_e64 v0, -s6
+; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-SDAG-NEXT: s_endpgm
+;
+; SI-GISEL-LABEL: cvt_flr_i32_f32_fneg:
+; SI-GISEL: ; %bb.0:
+; SI-GISEL-NEXT: s_load_dword s3, s[4:5], 0xb
+; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-GISEL-NEXT: s_mov_b32 s2, -1
+; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; SI-GISEL-NEXT: v_mul_f32_e64 v0, 1.0, -s3
+; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
+; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-GISEL-NEXT: s_endpgm
+;
+; GFX11-SDAG-LABEL: cvt_flr_i32_f32_fneg:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_clause 0x1
+; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-SDAG-NEXT: v_floor_f32_e64 v0, -s2
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: s_endpgm
+;
+; GFX11-GISEL-LABEL: cvt_flr_i32_f32_fneg:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_clause 0x1
+; GFX11-GISEL-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_max_f32_e64 v0, -s2, -s2
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-GISEL-NEXT: s_endpgm
%x.fneg = fsub float -0.000000e+00, %x
- %floor = call float @llvm.floor.f32(float %x.fneg) #1
+ %floor = call nnan float @llvm.floor.f32(float %x.fneg) #1
%cvt = fptosi float %floor to i32
store i32 %cvt, ptr addrspace(1) %out
ret void
}
-; FUNC-LABEL: {{^}}cvt_flr_i32_f32_fabs_fneg:
-; SI-NOT: add
-; SI-SAFE-NOT: v_cvt_flr_i32_f32
-; SI-NONAN: v_cvt_flr_i32_f32_e64 v{{[0-9]+}}, -|s{{[0-9]+}}|
-; SI: s_endpgm
define amdgpu_kernel void @cvt_flr_i32_f32_fabs_fneg(ptr addrspace(1) %out, float %x) #0 {
+; SI-SDAG-LABEL: cvt_flr_i32_f32_fabs_fneg:
+; SI-SDAG: ; %bb.0:
+; SI-SDAG-NEXT: s_load_dword s6, s[4:5], 0xb
+; SI-SDAG-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
+; SI-SDAG-NEXT: s_mov_b32 s2, -1
+; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; SI-SDAG-NEXT: v_floor_f32_e64 v0, -|s6|
+; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-SDAG-NEXT: s_endpgm
+;
+; SI-GISEL-LABEL: cvt_flr_i32_f32_fabs_fneg:
+; SI-GISEL: ; %bb.0:
+; SI-GISEL-NEXT: s_load_dword s3, s[4:5], 0xb
+; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-GISEL-NEXT: s_mov_b32 s2, -1
+; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; SI-GISEL-NEXT: v_mul_f32_e64 v0, 1.0, -|s3|
+; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
+; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-GISEL-NEXT: s_endpgm
+;
+; GFX11-SDAG-LABEL: cvt_flr_i32_f32_fabs_fneg:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_clause 0x1
+; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-SDAG-NEXT: v_floor_f32_e64 v0, -|s2|
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: s_endpgm
+;
+; GFX11-GISEL-LABEL: cvt_flr_i32_f32_fabs_fneg:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_clause 0x1
+; GFX11-GISEL-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_max_f32_e64 v0, -|s2|, -|s2|
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-GISEL-NEXT: s_endpgm
%x.fabs = call float @llvm.fabs.f32(float %x) #1
%x.fabs.fneg = fsub float -0.000000e+00, %x.fabs
- %floor = call float @llvm.floor.f32(float %x.fabs.fneg) #1
+ %floor = call nnan float @llvm.floor.f32(float %x.fabs.fneg) #1
%cvt = fptosi float %floor to i32
store i32 %cvt, ptr addrspace(1) %out
ret void
}
-; FUNC-LABEL: {{^}}no_cvt_flr_i32_f32_0:
-; SI-NOT: v_cvt_flr_i32_f32
-; SI: v_floor_f32
-; SI: v_cvt_u32_f32_e32
-; SI: s_endpgm
define amdgpu_kernel void @no_cvt_flr_i32_f32_0(ptr addrspace(1) %out, float %x) #0 {
- %floor = call float @llvm.floor.f32(float %x) #1
+;
+; SI-SDAG-LABEL: no_cvt_flr_i32_f32_0:
+; SI-SDAG: ; %bb.0:
+; SI-SDAG-NEXT: s_load_dword s6, s[4:5], 0xb
+; SI-SDAG-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
+; SI-SDAG-NEXT: s_mov_b32 s2, -1
+; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; SI-SDAG-NEXT: v_floor_f32_e32 v0, s6
+; SI-SDAG-NEXT: v_cvt_u32_f32_e32 v0, v0
+; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-SDAG-NEXT: s_endpgm
+;
+; SI-GISEL-LABEL: no_cvt_flr_i32_f32_0:
+; SI-GISEL: ; %bb.0:
+; SI-GISEL-NEXT: s_load_dword s3, s[4:5], 0xb
+; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-GISEL-NEXT: s_mov_b32 s2, -1
+; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; SI-GISEL-NEXT: v_floor_f32_e32 v0, s3
+; SI-GISEL-NEXT: v_cvt_u32_f32_e32 v0, v0
+; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
+; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-GISEL-NEXT: s_endpgm
+;
+; GFX11-SDAG-LABEL: no_cvt_flr_i32_f32_0:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_clause 0x1
+; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, s2
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_cvt_u32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: s_endpgm
+;
+; GFX11-GISEL-LABEL: no_cvt_flr_i32_f32_0:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_clause 0x1
+; GFX11-GISEL-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, s2
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_cvt_u32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-GISEL-NEXT: s_endpgm
+ %floor = call nnan float @llvm.floor.f32(float %x) #1
%cvt = fptoui float %floor to i32
store i32 %cvt, ptr addrspace(1) %out
ret void
diff --git a/llvm/test/CodeGen/AMDGPU/cvt_rpi_i32_f32.ll b/llvm/test/CodeGen/AMDGPU/cvt_rpi_i32_f32.ll
index 0203b2d4f896f..f1ad8ba078a06 100644
--- a/llvm/test/CodeGen/AMDGPU/cvt_rpi_i32_f32.ll
+++ b/llvm/test/CodeGen/AMDGPU/cvt_rpi_i32_f32.ll
@@ -1,79 +1,325 @@
-; RUN: llc -mtriple=amdgcn < %s | FileCheck -check-prefix=SI-SAFE -check-prefix=SI -check-prefix=FUNC %s
-; RUN: llc -mtriple=amdgcn -enable-no-nans-fp-math < %s | FileCheck -check-prefix=SI-NONAN -check-prefix=SI -check-prefix=FUNC %s
-; RUN: llc -mtriple=amdgcn -mcpu=tonga < %s | FileCheck -check-prefix=SI-SAFE -check-prefix=SI -check-prefix=FUNC %s
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgcn < %s | FileCheck -check-prefix=SI-SDAG %s
+; RUN: llc -mtriple=amdgcn -global-isel < %s | FileCheck -check-prefix=SI-GISEL %s
+; RUN: llc -mtriple=amdgcn -mcpu=gfx1100 < %s | FileCheck -check-prefix=GFX11-SDAG %s
+; RUN: llc -mtriple=amdgcn -mcpu=gfx1100 -global-isel < %s | FileCheck -check-prefix=GFX11-GISEL %s
declare float @llvm.fabs.f32(float) #1
declare float @llvm.floor.f32(float) #1
-; FUNC-LABEL: {{^}}cvt_rpi_i32_f32:
-; SI-SAFE-NOT: v_cvt_rpi_i32_f32
-; SI-NONAN: v_cvt_rpi_i32_f32_e32 v{{[0-9]+}}, s{{[0-9]+}}
-; SI: s_endpgm
define amdgpu_kernel void @cvt_rpi_i32_f32(ptr addrspace(1) %out, float %x) #0 {
+; SI-SDAG-LABEL: cvt_rpi_i32_f32:
+; SI-SDAG: ; %bb.0:
+; SI-SDAG-NEXT: s_load_dword s6, s[4:5], 0xb
+; SI-SDAG-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
+; SI-SDAG-NEXT: s_mov_b32 s2, -1
+; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; SI-SDAG-NEXT: v_add_f32_e64 v0, s6, 0.5
+; SI-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-SDAG-NEXT: s_endpgm
+;
+; SI-GISEL-LABEL: cvt_rpi_i32_f32:
+; SI-GISEL: ; %bb.0:
+; SI-GISEL-NEXT: s_load_dword s3, s[4:5], 0xb
+; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-GISEL-NEXT: s_mov_b32 s2, -1
+; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; SI-GISEL-NEXT: v_add_f32_e64 v0, s3, 0.5
+; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
+; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-GISEL-NEXT: s_endpgm
+;
+; GFX11-SDAG-LABEL: cvt_rpi_i32_f32:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_clause 0x1
+; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-SDAG-NEXT: v_add_f32_e64 v0, s2, 0.5
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: s_endpgm
+;
+; GFX11-GISEL-LABEL: cvt_rpi_i32_f32:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_clause 0x1
+; GFX11-GISEL-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_add_f32_e64 v0, s2, 0.5
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-GISEL-NEXT: s_endpgm
%fadd = fadd float %x, 0.5
- %floor = call float @llvm.floor.f32(float %fadd) #1
+ %floor = call nnan float @llvm.floor.f32(float %fadd) #1
%cvt = fptosi float %floor to i32
store i32 %cvt, ptr addrspace(1) %out
ret void
}
-; FUNC-LABEL: {{^}}cvt_rpi_i32_f32_fabs:
-; SI-SAFE-NOT: v_cvt_rpi_i32_f32
-; SI-NONAN: v_cvt_rpi_i32_f32_e64 v{{[0-9]+}}, |s{{[0-9]+}}|{{$}}
-; SI: s_endpgm
define amdgpu_kernel void @cvt_rpi_i32_f32_fabs(ptr addrspace(1) %out, float %x) #0 {
+; SI-SDAG-LABEL: cvt_rpi_i32_f32_fabs:
+; SI-SDAG: ; %bb.0:
+; SI-SDAG-NEXT: s_load_dword s6, s[4:5], 0xb
+; SI-SDAG-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
+; SI-SDAG-NEXT: s_mov_b32 s2, -1
+; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; SI-SDAG-NEXT: v_add_f32_e64 v0, |s6|, 0.5
+; SI-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-SDAG-NEXT: s_endpgm
+;
+; SI-GISEL-LABEL: cvt_rpi_i32_f32_fabs:
+; SI-GISEL: ; %bb.0:
+; SI-GISEL-NEXT: s_load_dword s3, s[4:5], 0xb
+; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-GISEL-NEXT: s_mov_b32 s2, -1
+; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; SI-GISEL-NEXT: v_add_f32_e64 v0, |s3|, 0.5
+; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
+; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-GISEL-NEXT: s_endpgm
+;
+; GFX11-SDAG-LABEL: cvt_rpi_i32_f32_fabs:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_clause 0x1
+; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-SDAG-NEXT: v_add_f32_e64 v0, |s2|, 0.5
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: s_endpgm
+;
+; GFX11-GISEL-LABEL: cvt_rpi_i32_f32_fabs:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_clause 0x1
+; GFX11-GISEL-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_add_f32_e64 v0, |s2|, 0.5
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-GISEL-NEXT: s_endpgm
%x.fabs = call float @llvm.fabs.f32(float %x) #1
%fadd = fadd float %x.fabs, 0.5
- %floor = call float @llvm.floor.f32(float %fadd) #1
+ %floor = call nnan float @llvm.floor.f32(float %fadd) #1
%cvt = fptosi float %floor to i32
store i32 %cvt, ptr addrspace(1) %out
ret void
}
; FIXME: This doesn't work because it forms fsub 0.5, x
-; FUNC-LABEL: {{^}}cvt_rpi_i32_f32_fneg:
-; XSI-NONAN: v_cvt_rpi_i32_f32_e64 v{{[0-9]+}}, -s{{[0-9]+}}
-; SI: v_sub_f32_e64 [[TMP:v[0-9]+]], 0.5, s{{[0-9]+}}
-; SI-SAFE-NOT: v_cvt_flr_i32_f32
-; SI-NONAN: v_cvt_flr_i32_f32_e32 {{v[0-9]+}}, [[TMP]]
-; SI: s_endpgm
define amdgpu_kernel void @cvt_rpi_i32_f32_fneg(ptr addrspace(1) %out, float %x) #0 {
+; SI-SDAG-LABEL: cvt_rpi_i32_f32_fneg:
+; SI-SDAG: ; %bb.0:
+; SI-SDAG-NEXT: s_load_dword s6, s[4:5], 0xb
+; SI-SDAG-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
+; SI-SDAG-NEXT: s_mov_b32 s2, -1
+; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; SI-SDAG-NEXT: v_sub_f32_e64 v0, 0.5, s6
+; SI-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-SDAG-NEXT: s_endpgm
+;
+; SI-GISEL-LABEL: cvt_rpi_i32_f32_fneg:
+; SI-GISEL: ; %bb.0:
+; SI-GISEL-NEXT: s_load_dword s3, s[4:5], 0xb
+; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-GISEL-NEXT: s_mov_b32 s2, -1
+; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; SI-GISEL-NEXT: v_mul_f32_e64 v0, 1.0, -s3
+; SI-GISEL-NEXT: v_add_f32_e32 v0, 0.5, v0
+; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
+; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-GISEL-NEXT: s_endpgm
+;
+; GFX11-SDAG-LABEL: cvt_rpi_i32_f32_fneg:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_clause 0x1
+; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-SDAG-NEXT: v_sub_f32_e64 v0, 0.5, s2
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: s_endpgm
+;
+; GFX11-GISEL-LABEL: cvt_rpi_i32_f32_fneg:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_load_b32 s0, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_max_f32_e64 v0, -s0, -s0
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_add_f32_e32 v0, 0.5, v0
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-GISEL-NEXT: s_endpgm
%x.fneg = fsub float -0.000000e+00, %x
%fadd = fadd float %x.fneg, 0.5
- %floor = call float @llvm.floor.f32(float %fadd) #1
+ %floor = call nnan float @llvm.floor.f32(float %fadd) #1
%cvt = fptosi float %floor to i32
store i32 %cvt, ptr addrspace(1) %out
ret void
}
; FIXME: This doesn't work for same reason as above
-; FUNC-LABEL: {{^}}cvt_rpi_i32_f32_fabs_fneg:
-; SI-SAFE-NOT: v_cvt_rpi_i32_f32
-; XSI-NONAN: v_cvt_rpi_i32_f32_e64 v{{[0-9]+}}, -|s{{[0-9]+}}|
-
-; SI: v_sub_f32_e64 [[TMP:v[0-9]+]], 0.5, |s{{[0-9]+}}|
-; SI-SAFE-NOT: v_cvt_flr_i32_f32
-; SI-NONAN: v_cvt_flr_i32_f32_e32 {{v[0-9]+}}, [[TMP]]
-; SI: s_endpgm
define amdgpu_kernel void @cvt_rpi_i32_f32_fabs_fneg(ptr addrspace(1) %out, float %x) #0 {
+;
+; SI-SDAG-LABEL: cvt_rpi_i32_f32_fabs_fneg:
+; SI-SDAG: ; %bb.0:
+; SI-SDAG-NEXT: s_load_dword s6, s[4:5], 0xb
+; SI-SDAG-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
+; SI-SDAG-NEXT: s_mov_b32 s2, -1
+; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; SI-SDAG-NEXT: v_sub_f32_e64 v0, 0.5, |s6|
+; SI-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-SDAG-NEXT: s_endpgm
+;
+; SI-GISEL-LABEL: cvt_rpi_i32_f32_fabs_fneg:
+; SI-GISEL: ; %bb.0:
+; SI-GISEL-NEXT: s_load_dword s3, s[4:5], 0xb
+; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-GISEL-NEXT: s_mov_b32 s2, -1
+; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; SI-GISEL-NEXT: v_mul_f32_e64 v0, 1.0, -|s3|
+; SI-GISEL-NEXT: v_add_f32_e32 v0, 0.5, v0
+; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
+; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-GISEL-NEXT: s_endpgm
+;
+; GFX11-SDAG-LABEL: cvt_rpi_i32_f32_fabs_fneg:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_clause 0x1
+; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-SDAG-NEXT: v_sub_f32_e64 v0, 0.5, |s2|
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: s_endpgm
+;
+; GFX11-GISEL-LABEL: cvt_rpi_i32_f32_fabs_fneg:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_load_b32 s0, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_max_f32_e64 v0, -|s0|, -|s0|
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_add_f32_e32 v0, 0.5, v0
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-GISEL-NEXT: s_endpgm
%x.fabs = call float @llvm.fabs.f32(float %x) #1
%x.fabs.fneg = fsub float -0.000000e+00, %x.fabs
%fadd = fadd float %x.fabs.fneg, 0.5
- %floor = call float @llvm.floor.f32(float %fadd) #1
+ %floor = call nnan float @llvm.floor.f32(float %fadd) #1
%cvt = fptosi float %floor to i32
store i32 %cvt, ptr addrspace(1) %out
ret void
}
-; FUNC-LABEL: {{^}}no_cvt_rpi_i32_f32_0:
-; SI-NOT: v_cvt_rpi_i32_f32
-; SI: v_add_f32
-; SI: v_floor_f32
-; SI: v_cvt_u32_f32
-; SI: s_endpgm
define amdgpu_kernel void @no_cvt_rpi_i32_f32_0(ptr addrspace(1) %out, float %x) #0 {
+; SI-SDAG-LABEL: no_cvt_rpi_i32_f32_0:
+; SI-SDAG: ; %bb.0:
+; SI-SDAG-NEXT: s_load_dword s6, s[4:5], 0xb
+; SI-SDAG-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
+; SI-SDAG-NEXT: s_mov_b32 s2, -1
+; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; SI-SDAG-NEXT: v_add_f32_e64 v0, s6, 0.5
+; SI-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_u32_f32_e32 v0, v0
+; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-SDAG-NEXT: s_endpgm
+;
+; SI-GISEL-LABEL: no_cvt_rpi_i32_f32_0:
+; SI-GISEL: ; %bb.0:
+; SI-GISEL-NEXT: s_load_dword s3, s[4:5], 0xb
+; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
+; SI-GISEL-NEXT: s_mov_b32 s2, -1
+; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; SI-GISEL-NEXT: v_add_f32_e64 v0, s3, 0.5
+; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_u32_f32_e32 v0, v0
+; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
+; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
+; SI-GISEL-NEXT: s_endpgm
+;
+; GFX11-SDAG-LABEL: no_cvt_rpi_i32_f32_0:
+; GFX11-SDAG: ; %bb.0:
+; GFX11-SDAG-NEXT: s_clause 0x1
+; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-SDAG-NEXT: v_add_f32_e64 v0, s2, 0.5
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: v_cvt_u32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: s_endpgm
+;
+; GFX11-GISEL-LABEL: no_cvt_rpi_i32_f32_0:
+; GFX11-GISEL: ; %bb.0:
+; GFX11-GISEL-NEXT: s_clause 0x1
+; GFX11-GISEL-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
+; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_add_f32_e64 v0, s2, 0.5
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_cvt_u32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-GISEL-NEXT: s_endpgm
%fadd = fadd float %x, 0.5
- %floor = call float @llvm.floor.f32(float %fadd) #1
+ %floor = call nnan float @llvm.floor.f32(float %fadd) #1
%cvt = fptoui float %floor to i32
store i32 %cvt, ptr addrspace(1) %out
ret void
>From 5f33df14a10a095e9d11215ecb95f6509c98930c Mon Sep 17 00:00:00 2001
From: Mirko Brkusanin <Mirko.Brkusanin at amd.com>
Date: Mon, 26 Jan 2026 16:00:18 +0100
Subject: [PATCH 2/2] [AMDGPU] Update patterns for v_cvt_flr and v_cvt_rpi
Support GlobalISel and switch to checking `nnan` flag on instruction
instead of TargetOptions.
Instruction are renamed to v_cvt_floor and v_cvt_nearest on gfx11+
so add gfx11 tests as well.
---
.../include/llvm/Target/TargetSelectionDAG.td | 8 ++
llvm/lib/Target/AMDGPU/AMDGPUInstructions.td | 20 ++--
llvm/test/CodeGen/AMDGPU/cvt_flr_i32_f32.ll | 91 +++++++-----------
llvm/test/CodeGen/AMDGPU/cvt_rpi_i32_f32.ll | 94 ++++++-------------
4 files changed, 82 insertions(+), 131 deletions(-)
diff --git a/llvm/include/llvm/Target/TargetSelectionDAG.td b/llvm/include/llvm/Target/TargetSelectionDAG.td
index bcb4500942111..b297fd06711a5 100644
--- a/llvm/include/llvm/Target/TargetSelectionDAG.td
+++ b/llvm/include/llvm/Target/TargetSelectionDAG.td
@@ -1201,6 +1201,14 @@ def sext_like : PatFrags<(ops node:$src),
[(zext_nneg node:$src),
(sext node:$src)]>;
+def ffloor_nnan : PatFrag<(ops node:$src), (ffloor node:$src), [{
+ return N->getFlags().hasNoNaNs();
+}]> {
+ let GISelPredicateCode = [{
+ return MI.getFlag(MachineInstr::FmNoNans);
+ }];
+}
+
// null_frag - The null pattern operator is used in multiclass instantiations
// which accept an SDPatternOperator for use in matching patterns for internal
// definitions. When expanding a pattern, if the null fragment is referenced
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUInstructions.td b/llvm/lib/Target/AMDGPU/AMDGPUInstructions.td
index 2a99dacba52a4..2d649c2b7c5eb 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUInstructions.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPUInstructions.td
@@ -758,8 +758,11 @@ def FP_ONE : PatLeaf <
def FP_HALF : PatLeaf <
(fpimm),
- [{return N->isExactlyValue(0.5);}]
->;
+ [{return N->isExactlyValue(0.5);}]> {
+ let GISelPredicateCode = [{
+ return MI.getOperand(1).getFPImm()->isExactlyValue(0.5);
+ }];
+}
/* Generic helper patterns for intrinsics */
/* -------------------------------------- */
@@ -806,16 +809,15 @@ class DwordAddrPat<ValueType vt, RegisterClass rc> : AMDGPUPat <
// Special conversion patterns
-def cvt_rpi_i32_f32 : PatFrag <
+let GIIgnoreCopies = 1 in
+def cvt_rpi_i32_f32 : PatFrag<
(ops node:$src),
- (fp_to_sint (ffloor (fadd $src, FP_HALF))),
- [{ (void) N; return TM.Options.NoNaNsFPMath; }]
->;
+ (fp_to_sint (ffloor_nnan (fadd $src, FP_HALF)))
+>, GISelFlags;
-def cvt_flr_i32_f32 : PatFrag <
+def cvt_flr_i32_f32 : PatFrag<
(ops node:$src),
- (fp_to_sint (ffloor $src)),
- [{ (void)N; return TM.Options.NoNaNsFPMath; }]
+ (fp_to_sint (ffloor_nnan $src))
>;
let AddedComplexity = 2 in {
diff --git a/llvm/test/CodeGen/AMDGPU/cvt_flr_i32_f32.ll b/llvm/test/CodeGen/AMDGPU/cvt_flr_i32_f32.ll
index dcba86b04a934..9592e39114ea8 100644
--- a/llvm/test/CodeGen/AMDGPU/cvt_flr_i32_f32.ll
+++ b/llvm/test/CodeGen/AMDGPU/cvt_flr_i32_f32.ll
@@ -15,8 +15,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_0(ptr addrspace(1) %out, float %x) #0
; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
; SI-SDAG-NEXT: s_mov_b32 s2, -1
; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; SI-SDAG-NEXT: v_floor_f32_e32 v0, s6
-; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_flr_i32_f32_e32 v0, s6
; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-SDAG-NEXT: s_endpgm
;
@@ -26,8 +25,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_0(ptr addrspace(1) %out, float %x) #0
; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; SI-GISEL-NEXT: s_mov_b32 s2, -1
; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
-; SI-GISEL-NEXT: v_floor_f32_e32 v0, s3
-; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_flr_i32_f32_e32 v0, s3
; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-GISEL-NEXT: s_endpgm
@@ -37,12 +35,10 @@ define amdgpu_kernel void @cvt_flr_i32_f32_0(ptr addrspace(1) %out, float %x) #0
; GFX11-SDAG-NEXT: s_clause 0x1
; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, s2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: v_cvt_floor_i32_f32_e32 v1, s2
+; GFX11-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-SDAG-NEXT: s_endpgm
;
; GFX11-GISEL-LABEL: cvt_flr_i32_f32_0:
@@ -52,9 +48,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_0(ptr addrspace(1) %out, float %x) #0
; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, s2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_cvt_floor_i32_f32_e32 v0, s2
; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-GISEL-NEXT: s_endpgm
%floor = call nnan float @llvm.floor.f32(float %x) #1
@@ -72,8 +66,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_1(ptr addrspace(1) %out, float %x) #0
; SI-SDAG-NEXT: s_mov_b32 s2, -1
; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; SI-SDAG-NEXT: v_add_f32_e64 v0, s6, 1.0
-; SI-SDAG-NEXT: v_floor_f32_e32 v0, v0
-; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_flr_i32_f32_e32 v0, v0
; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-SDAG-NEXT: s_endpgm
;
@@ -84,8 +77,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_1(ptr addrspace(1) %out, float %x) #0
; SI-GISEL-NEXT: s_mov_b32 s2, -1
; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; SI-GISEL-NEXT: v_add_f32_e64 v0, s3, 1.0
-; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_flr_i32_f32_e32 v0, v0
; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-GISEL-NEXT: s_endpgm
@@ -98,9 +90,8 @@ define amdgpu_kernel void @cvt_flr_i32_f32_1(ptr addrspace(1) %out, float %x) #0
; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-NEXT: v_add_f32_e64 v0, s2, 1.0
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_cvt_floor_i32_f32_e32 v0, v0
; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-SDAG-NEXT: s_endpgm
;
@@ -112,9 +103,8 @@ define amdgpu_kernel void @cvt_flr_i32_f32_1(ptr addrspace(1) %out, float %x) #0
; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_add_f32_e64 v0, s2, 1.0
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_cvt_floor_i32_f32_e32 v0, v0
; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-GISEL-NEXT: s_endpgm
%fadd = fadd float %x, 1.0
@@ -132,8 +122,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fabs(ptr addrspace(1) %out, float %x)
; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
; SI-SDAG-NEXT: s_mov_b32 s2, -1
; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; SI-SDAG-NEXT: v_floor_f32_e64 v0, |s6|
-; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_flr_i32_f32_e64 v0, |s6|
; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-SDAG-NEXT: s_endpgm
;
@@ -143,8 +132,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fabs(ptr addrspace(1) %out, float %x)
; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; SI-GISEL-NEXT: s_mov_b32 s2, -1
; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
-; SI-GISEL-NEXT: v_floor_f32_e64 v0, |s3|
-; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_flr_i32_f32_e64 v0, |s3|
; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-GISEL-NEXT: s_endpgm
@@ -154,12 +142,10 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fabs(ptr addrspace(1) %out, float %x)
; GFX11-SDAG-NEXT: s_clause 0x1
; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-SDAG-NEXT: v_floor_f32_e64 v0, |s2|
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: v_cvt_floor_i32_f32_e64 v1, |s2|
+; GFX11-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-SDAG-NEXT: s_endpgm
;
; GFX11-GISEL-LABEL: cvt_flr_i32_f32_fabs:
@@ -169,9 +155,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fabs(ptr addrspace(1) %out, float %x)
; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-GISEL-NEXT: v_floor_f32_e64 v0, |s2|
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_cvt_floor_i32_f32_e64 v0, |s2|
; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-GISEL-NEXT: s_endpgm
%x.fabs = call float @llvm.fabs.f32(float %x) #1
@@ -181,6 +165,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fabs(ptr addrspace(1) %out, float %x)
ret void
}
+; FIXME: GlobalISel selecting modifier fails because of G_FCANONICALIZE
define amdgpu_kernel void @cvt_flr_i32_f32_fneg(ptr addrspace(1) %out, float %x) #0 {
; SI-SDAG-LABEL: cvt_flr_i32_f32_fneg:
; SI-SDAG: ; %bb.0:
@@ -189,8 +174,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fneg(ptr addrspace(1) %out, float %x)
; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
; SI-SDAG-NEXT: s_mov_b32 s2, -1
; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; SI-SDAG-NEXT: v_floor_f32_e64 v0, -s6
-; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_flr_i32_f32_e64 v0, -s6
; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-SDAG-NEXT: s_endpgm
;
@@ -201,8 +185,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fneg(ptr addrspace(1) %out, float %x)
; SI-GISEL-NEXT: s_mov_b32 s2, -1
; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; SI-GISEL-NEXT: v_mul_f32_e64 v0, 1.0, -s3
-; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_flr_i32_f32_e32 v0, v0
; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-GISEL-NEXT: s_endpgm
@@ -212,12 +195,10 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fneg(ptr addrspace(1) %out, float %x)
; GFX11-SDAG-NEXT: s_clause 0x1
; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-SDAG-NEXT: v_floor_f32_e64 v0, -s2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: v_cvt_floor_i32_f32_e64 v1, -s2
+; GFX11-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-SDAG-NEXT: s_endpgm
;
; GFX11-GISEL-LABEL: cvt_flr_i32_f32_fneg:
@@ -228,9 +209,8 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fneg(ptr addrspace(1) %out, float %x)
; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_f32_e64 v0, -s2, -s2
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_cvt_floor_i32_f32_e32 v0, v0
; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-GISEL-NEXT: s_endpgm
%x.fneg = fsub float -0.000000e+00, %x
@@ -248,8 +228,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fabs_fneg(ptr addrspace(1) %out, floa
; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
; SI-SDAG-NEXT: s_mov_b32 s2, -1
; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; SI-SDAG-NEXT: v_floor_f32_e64 v0, -|s6|
-; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_flr_i32_f32_e64 v0, -|s6|
; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-SDAG-NEXT: s_endpgm
;
@@ -260,8 +239,7 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fabs_fneg(ptr addrspace(1) %out, floa
; SI-GISEL-NEXT: s_mov_b32 s2, -1
; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; SI-GISEL-NEXT: v_mul_f32_e64 v0, 1.0, -|s3|
-; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_flr_i32_f32_e32 v0, v0
; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-GISEL-NEXT: s_endpgm
@@ -271,12 +249,10 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fabs_fneg(ptr addrspace(1) %out, floa
; GFX11-SDAG-NEXT: s_clause 0x1
; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-SDAG-NEXT: v_floor_f32_e64 v0, -|s2|
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: v_cvt_floor_i32_f32_e64 v1, -|s2|
+; GFX11-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-SDAG-NEXT: s_endpgm
;
; GFX11-GISEL-LABEL: cvt_flr_i32_f32_fabs_fneg:
@@ -287,9 +263,8 @@ define amdgpu_kernel void @cvt_flr_i32_f32_fabs_fneg(ptr addrspace(1) %out, floa
; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-GISEL-NEXT: v_max_f32_e64 v0, -|s2|, -|s2|
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-GISEL-NEXT: v_cvt_floor_i32_f32_e32 v0, v0
; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-GISEL-NEXT: s_endpgm
%x.fabs = call float @llvm.fabs.f32(float %x) #1
diff --git a/llvm/test/CodeGen/AMDGPU/cvt_rpi_i32_f32.ll b/llvm/test/CodeGen/AMDGPU/cvt_rpi_i32_f32.ll
index f1ad8ba078a06..95d2fe3630671 100644
--- a/llvm/test/CodeGen/AMDGPU/cvt_rpi_i32_f32.ll
+++ b/llvm/test/CodeGen/AMDGPU/cvt_rpi_i32_f32.ll
@@ -15,9 +15,7 @@ define amdgpu_kernel void @cvt_rpi_i32_f32(ptr addrspace(1) %out, float %x) #0 {
; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
; SI-SDAG-NEXT: s_mov_b32 s2, -1
; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; SI-SDAG-NEXT: v_add_f32_e64 v0, s6, 0.5
-; SI-SDAG-NEXT: v_floor_f32_e32 v0, v0
-; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_rpi_i32_f32_e32 v0, s6
; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-SDAG-NEXT: s_endpgm
;
@@ -27,9 +25,7 @@ define amdgpu_kernel void @cvt_rpi_i32_f32(ptr addrspace(1) %out, float %x) #0 {
; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; SI-GISEL-NEXT: s_mov_b32 s2, -1
; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
-; SI-GISEL-NEXT: v_add_f32_e64 v0, s3, 0.5
-; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_rpi_i32_f32_e32 v0, s3
; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-GISEL-NEXT: s_endpgm
@@ -39,13 +35,10 @@ define amdgpu_kernel void @cvt_rpi_i32_f32(ptr addrspace(1) %out, float %x) #0 {
; GFX11-SDAG-NEXT: s_clause 0x1
; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-SDAG-NEXT: v_add_f32_e64 v0, s2, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: v_cvt_nearest_i32_f32_e32 v1, s2
+; GFX11-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-SDAG-NEXT: s_endpgm
;
; GFX11-GISEL-LABEL: cvt_rpi_i32_f32:
@@ -55,10 +48,7 @@ define amdgpu_kernel void @cvt_rpi_i32_f32(ptr addrspace(1) %out, float %x) #0 {
; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-GISEL-NEXT: v_add_f32_e64 v0, s2, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_cvt_nearest_i32_f32_e32 v0, s2
; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-GISEL-NEXT: s_endpgm
%fadd = fadd float %x, 0.5
@@ -76,9 +66,7 @@ define amdgpu_kernel void @cvt_rpi_i32_f32_fabs(ptr addrspace(1) %out, float %x)
; SI-SDAG-NEXT: s_mov_b32 s3, 0xf000
; SI-SDAG-NEXT: s_mov_b32 s2, -1
; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; SI-SDAG-NEXT: v_add_f32_e64 v0, |s6|, 0.5
-; SI-SDAG-NEXT: v_floor_f32_e32 v0, v0
-; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_rpi_i32_f32_e64 v0, |s6|
; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-SDAG-NEXT: s_endpgm
;
@@ -88,9 +76,7 @@ define amdgpu_kernel void @cvt_rpi_i32_f32_fabs(ptr addrspace(1) %out, float %x)
; SI-GISEL-NEXT: s_load_dwordx2 s[0:1], s[4:5], 0x9
; SI-GISEL-NEXT: s_mov_b32 s2, -1
; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
-; SI-GISEL-NEXT: v_add_f32_e64 v0, |s3|, 0.5
-; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_rpi_i32_f32_e64 v0, |s3|
; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-GISEL-NEXT: s_endpgm
@@ -100,13 +86,10 @@ define amdgpu_kernel void @cvt_rpi_i32_f32_fabs(ptr addrspace(1) %out, float %x)
; GFX11-SDAG-NEXT: s_clause 0x1
; GFX11-SDAG-NEXT: s_load_b32 s2, s[4:5], 0x2c
; GFX11-SDAG-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
+; GFX11-SDAG-NEXT: v_mov_b32_e32 v0, 0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-SDAG-NEXT: v_add_f32_e64 v0, |s2|, 0.5
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
+; GFX11-SDAG-NEXT: v_cvt_nearest_i32_f32_e64 v1, |s2|
+; GFX11-SDAG-NEXT: global_store_b32 v0, v1, s[0:1]
; GFX11-SDAG-NEXT: s_endpgm
;
; GFX11-GISEL-LABEL: cvt_rpi_i32_f32_fabs:
@@ -116,10 +99,7 @@ define amdgpu_kernel void @cvt_rpi_i32_f32_fabs(ptr addrspace(1) %out, float %x)
; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-GISEL-NEXT: v_add_f32_e64 v0, |s2|, 0.5
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_cvt_nearest_i32_f32_e64 v0, |s2|
; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-GISEL-NEXT: s_endpgm
%x.fabs = call float @llvm.fabs.f32(float %x) #1
@@ -140,8 +120,7 @@ define amdgpu_kernel void @cvt_rpi_i32_f32_fneg(ptr addrspace(1) %out, float %x)
; SI-SDAG-NEXT: s_mov_b32 s2, -1
; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; SI-SDAG-NEXT: v_sub_f32_e64 v0, 0.5, s6
-; SI-SDAG-NEXT: v_floor_f32_e32 v0, v0
-; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_flr_i32_f32_e32 v0, v0
; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-SDAG-NEXT: s_endpgm
;
@@ -152,9 +131,7 @@ define amdgpu_kernel void @cvt_rpi_i32_f32_fneg(ptr addrspace(1) %out, float %x)
; SI-GISEL-NEXT: s_mov_b32 s2, -1
; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; SI-GISEL-NEXT: v_mul_f32_e64 v0, 1.0, -s3
-; SI-GISEL-NEXT: v_add_f32_e32 v0, 0.5, v0
-; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_rpi_i32_f32_e32 v0, v0
; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-GISEL-NEXT: s_endpgm
@@ -167,25 +144,21 @@ define amdgpu_kernel void @cvt_rpi_i32_f32_fneg(ptr addrspace(1) %out, float %x)
; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-NEXT: v_sub_f32_e64 v0, 0.5, s2
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_cvt_floor_i32_f32_e32 v0, v0
; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-SDAG-NEXT: s_endpgm
;
; GFX11-GISEL-LABEL: cvt_rpi_i32_f32_fneg:
; GFX11-GISEL: ; %bb.0:
-; GFX11-GISEL-NEXT: s_load_b32 s0, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_clause 0x1
+; GFX11-GISEL-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-GISEL-NEXT: v_max_f32_e64 v0, -s0, -s0
-; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_add_f32_e32 v0, 0.5, v0
-; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_max_f32_e64 v0, -s2, -s2
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
-; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_cvt_nearest_i32_f32_e32 v0, v0
; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-GISEL-NEXT: s_endpgm
%x.fneg = fsub float -0.000000e+00, %x
@@ -207,8 +180,7 @@ define amdgpu_kernel void @cvt_rpi_i32_f32_fabs_fneg(ptr addrspace(1) %out, floa
; SI-SDAG-NEXT: s_mov_b32 s2, -1
; SI-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; SI-SDAG-NEXT: v_sub_f32_e64 v0, 0.5, |s6|
-; SI-SDAG-NEXT: v_floor_f32_e32 v0, v0
-; SI-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-SDAG-NEXT: v_cvt_flr_i32_f32_e32 v0, v0
; SI-SDAG-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-SDAG-NEXT: s_endpgm
;
@@ -219,9 +191,7 @@ define amdgpu_kernel void @cvt_rpi_i32_f32_fabs_fneg(ptr addrspace(1) %out, floa
; SI-GISEL-NEXT: s_mov_b32 s2, -1
; SI-GISEL-NEXT: s_waitcnt lgkmcnt(0)
; SI-GISEL-NEXT: v_mul_f32_e64 v0, 1.0, -|s3|
-; SI-GISEL-NEXT: v_add_f32_e32 v0, 0.5, v0
-; SI-GISEL-NEXT: v_floor_f32_e32 v0, v0
-; SI-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
+; SI-GISEL-NEXT: v_cvt_rpi_i32_f32_e32 v0, v0
; SI-GISEL-NEXT: s_mov_b32 s3, 0xf000
; SI-GISEL-NEXT: buffer_store_dword v0, off, s[0:3], 0
; SI-GISEL-NEXT: s_endpgm
@@ -234,25 +204,21 @@ define amdgpu_kernel void @cvt_rpi_i32_f32_fabs_fneg(ptr addrspace(1) %out, floa
; GFX11-SDAG-NEXT: v_mov_b32_e32 v1, 0
; GFX11-SDAG-NEXT: s_waitcnt lgkmcnt(0)
; GFX11-SDAG-NEXT: v_sub_f32_e64 v0, 0.5, |s2|
-; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-SDAG-NEXT: v_floor_f32_e32 v0, v0
-; GFX11-SDAG-NEXT: v_cvt_i32_f32_e32 v0, v0
+; GFX11-SDAG-NEXT: s_delay_alu instid0(VALU_DEP_1)
+; GFX11-SDAG-NEXT: v_cvt_floor_i32_f32_e32 v0, v0
; GFX11-SDAG-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-SDAG-NEXT: s_endpgm
;
; GFX11-GISEL-LABEL: cvt_rpi_i32_f32_fabs_fneg:
; GFX11-GISEL: ; %bb.0:
-; GFX11-GISEL-NEXT: s_load_b32 s0, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_clause 0x1
+; GFX11-GISEL-NEXT: s_load_b32 s2, s[4:5], 0x2c
+; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
; GFX11-GISEL-NEXT: v_mov_b32_e32 v1, 0
; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
-; GFX11-GISEL-NEXT: v_max_f32_e64 v0, -|s0|, -|s0|
-; GFX11-GISEL-NEXT: s_load_b64 s[0:1], s[4:5], 0x24
-; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1) | instskip(NEXT) | instid1(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_add_f32_e32 v0, 0.5, v0
-; GFX11-GISEL-NEXT: v_floor_f32_e32 v0, v0
+; GFX11-GISEL-NEXT: v_max_f32_e64 v0, -|s2|, -|s2|
; GFX11-GISEL-NEXT: s_delay_alu instid0(VALU_DEP_1)
-; GFX11-GISEL-NEXT: v_cvt_i32_f32_e32 v0, v0
-; GFX11-GISEL-NEXT: s_waitcnt lgkmcnt(0)
+; GFX11-GISEL-NEXT: v_cvt_nearest_i32_f32_e32 v0, v0
; GFX11-GISEL-NEXT: global_store_b32 v1, v0, s[0:1]
; GFX11-GISEL-NEXT: s_endpgm
%x.fabs = call float @llvm.fabs.f32(float %x) #1
More information about the llvm-commits
mailing list