[llvm] [NFC][AMDGPU] Add tests for fptrunc folded into V_MAD/FMA_MIX (PR #219723)

via llvm-commits llvm-commits at lists.llvm.org
Sat Aug 29 13:33:30 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-amdgpu

Author: Dmitry Sidorov (MrSidims)

<details>
<summary>Changes</summary>

Rounding an f32 multiply or multiply-add to f16 rounds twice, once to f32 and once to f16. The mix instructions round once, straight to f16, so they return a different value, for example on gfx90a for by one f16 ULP. Dropping that intermediate rounding is what contract permits.

Assisted-By: Claude Code Opus 5

---

Patch is 72.10 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/219723.diff


3 Files Affected:

- (added) llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-fp-contract-fast.ll (+121) 
- (added) llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-rounding-bf16.ll (+247) 
- (added) llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-rounding.ll (+1343) 


``````````diff
diff --git a/llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-fp-contract-fast.ll b/llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-fp-contract-fast.ll
new file mode 100644
index 0000000000000..ef6604d77a3e8
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-fp-contract-fast.ll
@@ -0,0 +1,121 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgpu9.00 -fp-contract=fast < %s | FileCheck -check-prefixes=GFX900 %s
+; RUN: llc -global-isel -mtriple=amdgpu9.00 -fp-contract=fast < %s | FileCheck -check-prefixes=GISEL-GFX900 %s
+; RUN: llc -mtriple=amdgpu12.50 -mattr=-real-true16 -fp-contract=fast < %s | FileCheck -check-prefixes=GFX1250 %s
+
+; Nothing here carries contract. -fp-contract=fast opts the whole function in,
+; so the mix instruction is still selected.
+
+define half @fp_contract_fast_fmul_to_f16(float %a, float %b) #0 {
+; GFX900-LABEL: fp_contract_fast_fmul_to_f16:
+; GFX900:       ; %bb.0:
+; GFX900-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX900-NEXT:    v_mad_mixlo_f16 v0, v0, v1, neg(0)
+; GFX900-NEXT:    s_setpc_b64 s[30:31]
+;
+; GISEL-GFX900-LABEL: fp_contract_fast_fmul_to_f16:
+; GISEL-GFX900:       ; %bb.0:
+; GISEL-GFX900-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GISEL-GFX900-NEXT:    v_mad_mixlo_f16 v0, v0, v1, neg(0)
+; GISEL-GFX900-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: fp_contract_fast_fmul_to_f16:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mixlo_f16 v0, v0, v1, neg(0)
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %mul = fmul float %a, %b
+  %cvt = fptrunc float %mul to half
+  ret half %cvt
+}
+
+define half @fp_contract_fast_fmuladd_to_f16(float %a, float %b, float %c) #0 {
+; GFX900-LABEL: fp_contract_fast_fmuladd_to_f16:
+; GFX900:       ; %bb.0:
+; GFX900-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX900-NEXT:    v_mad_mixlo_f16 v0, v0, v1, v2
+; GFX900-NEXT:    s_setpc_b64 s[30:31]
+;
+; GISEL-GFX900-LABEL: fp_contract_fast_fmuladd_to_f16:
+; GISEL-GFX900:       ; %bb.0:
+; GISEL-GFX900-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GISEL-GFX900-NEXT:    v_mad_mixlo_f16 v0, v0, v1, v2
+; GISEL-GFX900-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: fp_contract_fast_fmuladd_to_f16:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mixlo_f16 v0, v0, v1, v2
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %mad = call float @llvm.fmuladd.f32(float %a, float %b, float %c)
+  %cvt = fptrunc float %mad to half
+  ret half %cvt
+}
+
+define half @fp_contract_fast_fma_to_f16(float %a, float %b, float %c) #0 {
+; GFX900-LABEL: fp_contract_fast_fma_to_f16:
+; GFX900:       ; %bb.0:
+; GFX900-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX900-NEXT:    v_fma_f32 v0, v0, v1, v2
+; GFX900-NEXT:    v_cvt_f16_f32_e32 v0, v0
+; GFX900-NEXT:    s_setpc_b64 s[30:31]
+;
+; GISEL-GFX900-LABEL: fp_contract_fast_fma_to_f16:
+; GISEL-GFX900:       ; %bb.0:
+; GISEL-GFX900-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GISEL-GFX900-NEXT:    v_fma_f32 v0, v0, v1, v2
+; GISEL-GFX900-NEXT:    v_cvt_f16_f32_e32 v0, v0
+; GISEL-GFX900-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: fp_contract_fast_fma_to_f16:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mixlo_f16 v0, v0, v1, v2
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %fma = call float @llvm.fma.f32(float %a, float %b, float %c)
+  %cvt = fptrunc float %fma to half
+  ret half %cvt
+}
+
+define bfloat @fp_contract_fast_fmul_to_bf16(float %a, float %b) #0 {
+; GFX900-LABEL: fp_contract_fast_fmul_to_bf16:
+; GFX900:       ; %bb.0:
+; GFX900-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX900-NEXT:    v_mul_f32_e32 v0, v0, v1
+; GFX900-NEXT:    v_bfe_u32 v1, v0, 16, 1
+; GFX900-NEXT:    s_movk_i32 s4, 0x7fff
+; GFX900-NEXT:    v_add3_u32 v1, v1, v0, s4
+; GFX900-NEXT:    v_or_b32_e32 v2, 0x400000, v0
+; GFX900-NEXT:    v_cmp_u_f32_e32 vcc, v0, v0
+; GFX900-NEXT:    v_cndmask_b32_e32 v0, v1, v2, vcc
+; GFX900-NEXT:    v_lshrrev_b32_e32 v0, 16, v0
+; GFX900-NEXT:    s_setpc_b64 s[30:31]
+;
+; GISEL-GFX900-LABEL: fp_contract_fast_fmul_to_bf16:
+; GISEL-GFX900:       ; %bb.0:
+; GISEL-GFX900-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GISEL-GFX900-NEXT:    v_mul_f32_e32 v0, v0, v1
+; GISEL-GFX900-NEXT:    v_bfe_u32 v2, v0, 16, 1
+; GISEL-GFX900-NEXT:    v_mov_b32_e32 v3, 0x7fff
+; GISEL-GFX900-NEXT:    v_or_b32_e32 v1, 0x400000, v0
+; GISEL-GFX900-NEXT:    v_add3_u32 v2, v2, v0, v3
+; GISEL-GFX900-NEXT:    v_cmp_u_f32_e32 vcc, 0, v0
+; GISEL-GFX900-NEXT:    v_cndmask_b32_e32 v0, v2, v1, vcc
+; GISEL-GFX900-NEXT:    v_lshrrev_b32_e32 v0, 16, v0
+; GISEL-GFX900-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1250-LABEL: fp_contract_fast_fmul_to_bf16:
+; GFX1250:       ; %bb.0:
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mixlo_bf16 v0, v0, v1, neg(0)
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+  %mul = fmul float %a, %b
+  %cvt = fptrunc float %mul to bfloat
+  ret bfloat %cvt
+}
+
+attributes #0 = { nounwind denormal_fpenv(float: preservesign) }
diff --git a/llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-rounding-bf16.ll b/llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-rounding-bf16.ll
new file mode 100644
index 0000000000000..134d5cb97a9a8
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-rounding-bf16.ll
@@ -0,0 +1,247 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgpu12.50 -mattr=-real-true16 < %s | FileCheck -check-prefixes=GFX1250,GFX1250-FAKE16 %s
+; RUN: llc -mtriple=amdgpu12.50 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GFX1250,GFX1250-REAL16 %s
+
+; FIXME: the unflagged cases below are wrong. They select the mix
+; instruction and round once, dropping a rounding step the IR asks for.
+
+define bfloat @fptrunc_fmul_to_bf16(float %a, float %b) #0 {
+; GFX1250-LABEL: fptrunc_fmul_to_bf16:
+; GFX1250:       ; %bb.0: ; %.entry
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mixlo_bf16 v0, v0, v1, neg(0)
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %mul = fmul float %a, %b
+  %cvt = fptrunc float %mul to bfloat
+  ret bfloat %cvt
+}
+
+define bfloat @fptrunc_fmul_to_bf16_contract(float %a, float %b) #0 {
+; GFX1250-LABEL: fptrunc_fmul_to_bf16_contract:
+; GFX1250:       ; %bb.0: ; %.entry
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mixlo_bf16 v0, v0, v1, neg(0)
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %mul = fmul contract float %a, %b
+  %cvt = fptrunc contract float %mul to bfloat
+  ret bfloat %cvt
+}
+
+; The addend must be -0.0 so a product of -0.0 keeps its sign.
+define bfloat @fptrunc_fmul_by_zero_to_bf16_contract(float %a) #0 {
+; GFX1250-LABEL: fptrunc_fmul_by_zero_to_bf16_contract:
+; GFX1250:       ; %bb.0: ; %.entry
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mixlo_bf16 v0, v0, 0, neg(0)
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %mul = fmul contract float %a, 0.0
+  %cvt = fptrunc contract float %mul to bfloat
+  ret bfloat %cvt
+}
+
+define <2 x bfloat> @fptrunc_fmul_to_bf16_hi(float %a, float %b, bfloat %lo) #0 {
+; GFX1250-FAKE16-LABEL: fptrunc_fmul_to_bf16_hi:
+; GFX1250-FAKE16:       ; %bb.0: ; %.entry
+; GFX1250-FAKE16-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-FAKE16-NEXT:    v_fma_mixhi_bf16 v2, v0, v1, neg(0)
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v0, v2
+; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
+;
+; GFX1250-REAL16-LABEL: fptrunc_fmul_to_bf16_hi:
+; GFX1250-REAL16:       ; %bb.0: ; %.entry
+; GFX1250-REAL16-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-REAL16-NEXT:    v_fma_mixhi_bf16 v0, v0, v1, neg(0)
+; GFX1250-REAL16-NEXT:    v_mov_b16_e32 v0.l, v2.l
+; GFX1250-REAL16-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %mul = fmul float %a, %b
+  %cvt = fptrunc float %mul to bfloat
+  %v0 = insertelement <2 x bfloat> poison, bfloat %lo, i32 0
+  %v1 = insertelement <2 x bfloat> %v0, bfloat %cvt, i32 1
+  ret <2 x bfloat> %v1
+}
+
+define <2 x bfloat> @fptrunc_fmul_to_bf16_hi_contract(float %a, float %b, bfloat %lo) #0 {
+; GFX1250-FAKE16-LABEL: fptrunc_fmul_to_bf16_hi_contract:
+; GFX1250-FAKE16:       ; %bb.0: ; %.entry
+; GFX1250-FAKE16-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-FAKE16-NEXT:    v_fma_mixhi_bf16 v2, v0, v1, neg(0)
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v0, v2
+; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
+;
+; GFX1250-REAL16-LABEL: fptrunc_fmul_to_bf16_hi_contract:
+; GFX1250-REAL16:       ; %bb.0: ; %.entry
+; GFX1250-REAL16-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-REAL16-NEXT:    v_fma_mixhi_bf16 v0, v0, v1, neg(0)
+; GFX1250-REAL16-NEXT:    v_mov_b16_e32 v0.l, v2.l
+; GFX1250-REAL16-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %mul = fmul contract float %a, %b
+  %cvt = fptrunc contract float %mul to bfloat
+  %v0 = insertelement <2 x bfloat> poison, bfloat %lo, i32 0
+  %v1 = insertelement <2 x bfloat> %v0, bfloat %cvt, i32 1
+  ret <2 x bfloat> %v1
+}
+
+; Both multiplicands come from bf16, so the f32 product is exact and no
+; contract flag is needed.
+define bfloat @fptrunc_fmul_narrow_to_bf16(bfloat %a, bfloat %b) #0 {
+; GFX1250-LABEL: fptrunc_fmul_narrow_to_bf16:
+; GFX1250:       ; %bb.0: ; %.entry
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mixlo_bf16 v0, v0, v1, neg(0) op_sel_hi:[1,1,0]
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %a.ext = fpext bfloat %a to float
+  %b.ext = fpext bfloat %b to float
+  %mul = fmul float %a.ext, %b.ext
+  %cvt = fptrunc float %mul to bfloat
+  ret bfloat %cvt
+}
+
+define <2 x bfloat> @fptrunc_fmul_narrow_to_bf16_hi(bfloat %a, bfloat %b, bfloat %lo) #0 {
+; GFX1250-FAKE16-LABEL: fptrunc_fmul_narrow_to_bf16_hi:
+; GFX1250-FAKE16:       ; %bb.0: ; %.entry
+; GFX1250-FAKE16-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-FAKE16-NEXT:    v_fma_mixhi_bf16 v2, v0, v1, neg(0) op_sel_hi:[1,1,0]
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v0, v2
+; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
+;
+; GFX1250-REAL16-LABEL: fptrunc_fmul_narrow_to_bf16_hi:
+; GFX1250-REAL16:       ; %bb.0: ; %.entry
+; GFX1250-REAL16-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-REAL16-NEXT:    v_fma_mixhi_bf16 v0, v0, v1, neg(0) op_sel_hi:[1,1,0]
+; GFX1250-REAL16-NEXT:    v_mov_b16_e32 v0.l, v2.l
+; GFX1250-REAL16-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %a.ext = fpext bfloat %a to float
+  %b.ext = fpext bfloat %b to float
+  %mul = fmul float %a.ext, %b.ext
+  %cvt = fptrunc float %mul to bfloat
+  %v0 = insertelement <2 x bfloat> poison, bfloat %lo, i32 0
+  %v1 = insertelement <2 x bfloat> %v0, bfloat %cvt, i32 1
+  ret <2 x bfloat> %v1
+}
+
+; Only one multiplicand is narrow, so the double rounding is real.
+define bfloat @fptrunc_fmul_half_narrow_to_bf16(bfloat %a, float %b) #0 {
+; GFX1250-LABEL: fptrunc_fmul_half_narrow_to_bf16:
+; GFX1250:       ; %bb.0: ; %.entry
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mixlo_bf16 v0, v0, v1, neg(0) op_sel_hi:[1,0,0]
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %a.ext = fpext bfloat %a to float
+  %mul = fmul float %a.ext, %b
+  %cvt = fptrunc float %mul to bfloat
+  ret bfloat %cvt
+}
+
+define bfloat @fptrunc_fma_to_bf16(float %a, float %b, float %c) #0 {
+; GFX1250-LABEL: fptrunc_fma_to_bf16:
+; GFX1250:       ; %bb.0: ; %.entry
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mixlo_bf16 v0, v0, v1, v2
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %fma = call float @llvm.fma.f32(float %a, float %b, float %c)
+  %cvt = fptrunc float %fma to bfloat
+  ret bfloat %cvt
+}
+
+define bfloat @fptrunc_fma_to_bf16_contract(float %a, float %b, float %c) #0 {
+; GFX1250-LABEL: fptrunc_fma_to_bf16_contract:
+; GFX1250:       ; %bb.0: ; %.entry
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mixlo_bf16 v0, v0, v1, v2
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %fma = call contract float @llvm.fma.f32(float %a, float %b, float %c)
+  %cvt = fptrunc contract float %fma to bfloat
+  ret bfloat %cvt
+}
+
+define <2 x bfloat> @fptrunc_fma_to_bf16_hi(float %a, float %b, float %c, bfloat %lo) #0 {
+; GFX1250-FAKE16-LABEL: fptrunc_fma_to_bf16_hi:
+; GFX1250-FAKE16:       ; %bb.0: ; %.entry
+; GFX1250-FAKE16-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-FAKE16-NEXT:    v_fma_mixhi_bf16 v3, v0, v1, v2
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v0, v3
+; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
+;
+; GFX1250-REAL16-LABEL: fptrunc_fma_to_bf16_hi:
+; GFX1250-REAL16:       ; %bb.0: ; %.entry
+; GFX1250-REAL16-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-REAL16-NEXT:    v_fma_mixhi_bf16 v0, v0, v1, v2
+; GFX1250-REAL16-NEXT:    v_mov_b16_e32 v0.l, v3.l
+; GFX1250-REAL16-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %fma = call float @llvm.fma.f32(float %a, float %b, float %c)
+  %cvt = fptrunc float %fma to bfloat
+  %v0 = insertelement <2 x bfloat> poison, bfloat %lo, i32 0
+  %v1 = insertelement <2 x bfloat> %v0, bfloat %cvt, i32 1
+  ret <2 x bfloat> %v1
+}
+
+define <2 x bfloat> @fptrunc_fma_to_bf16_hi_contract(float %a, float %b, float %c, bfloat %lo) #0 {
+; GFX1250-FAKE16-LABEL: fptrunc_fma_to_bf16_hi_contract:
+; GFX1250-FAKE16:       ; %bb.0: ; %.entry
+; GFX1250-FAKE16-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-FAKE16-NEXT:    v_fma_mixhi_bf16 v3, v0, v1, v2
+; GFX1250-FAKE16-NEXT:    s_delay_alu instid0(VALU_DEP_1)
+; GFX1250-FAKE16-NEXT:    v_mov_b32_e32 v0, v3
+; GFX1250-FAKE16-NEXT:    s_set_pc_i64 s[30:31]
+;
+; GFX1250-REAL16-LABEL: fptrunc_fma_to_bf16_hi_contract:
+; GFX1250-REAL16:       ; %bb.0: ; %.entry
+; GFX1250-REAL16-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-REAL16-NEXT:    v_fma_mixhi_bf16 v0, v0, v1, v2
+; GFX1250-REAL16-NEXT:    v_mov_b16_e32 v0.l, v3.l
+; GFX1250-REAL16-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %fma = call contract float @llvm.fma.f32(float %a, float %b, float %c)
+  %cvt = fptrunc contract float %fma to bfloat
+  %v0 = insertelement <2 x bfloat> poison, bfloat %lo, i32 0
+  %v1 = insertelement <2 x bfloat> %v0, bfloat %cvt, i32 1
+  ret <2 x bfloat> %v1
+}
+
+; The f32 result is kept, so no contract flag is needed.
+define float @fmul_to_f32_neg_zero_addend(bfloat %a, float %b) #0 {
+; GFX1250-LABEL: fmul_to_f32_neg_zero_addend:
+; GFX1250:       ; %bb.0: ; %.entry
+; GFX1250-NEXT:    s_wait_loadcnt_dscnt 0x0
+; GFX1250-NEXT:    s_wait_kmcnt 0x0
+; GFX1250-NEXT:    v_fma_mix_f32_bf16 v0, v0, v1, neg(0) op_sel_hi:[1,0,0]
+; GFX1250-NEXT:    s_set_pc_i64 s[30:31]
+.entry:
+  %a.ext = fpext bfloat %a to float
+  %mul = fmul float %a.ext, %b
+  ret float %mul
+}
+
+attributes #0 = { nounwind denormal_fpenv(float: preservesign) }
diff --git a/llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-rounding.ll b/llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-rounding.ll
new file mode 100644
index 0000000000000..769d693985cbc
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/mad-mix-fptrunc-rounding.ll
@@ -0,0 +1,1343 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgpu8.03 < %s | FileCheck -check-prefixes=GFX803 %s
+; RUN: llc -mtriple=amdgpu9.00 < %s | FileCheck -check-prefixes=GFX900 %s
+; RUN: llc -mtriple=amdgpu9.06 < %s | FileCheck -check-prefixes=GFX906 %s
+; RUN: llc -mtriple=amdgpu9.0a < %s | FileCheck -check-prefixes=GFX90A %s
+; RUN: llc -mtriple=amdgpu11.00 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GFX1100 %s
+; RUN: llc -global-isel -mtriple=amdgpu9.00 < %s | FileCheck -check-prefixes=GISEL-GFX900 %s
+; RUN: llc -global-isel -mtriple=amdgpu9.06 < %s | FileCheck -check-prefixes=GISEL-GFX906 %s
+
+; V_{MAD,FMA}_MIX{LO,HI} round the f32 multiply or multiply-add once,
+; directly to f16, where the IR they match rounds to f32 first. On gfx90a
+; the two disagree for 5.95e-05 of random inputs, always by 1 f16 ULP, so
+; dropping the intermediate rounding needs contract on both the arithmetic
+; and the fptrunc. Every pattern is covered twice, unflagged and with
+; contract. gfx803 has no mix instructions. gfx900 uses V_MAD_MIX* and is
+; gated on NoFP32Denormals, hence the denormal_fpenv attribute below.
+;
+; FIXME: the unflagged cases below are wrong. They select the mix
+; instruction and round once, dropping a rounding step the IR asks for.
+
+define half @fptrunc_fmul_to_f16(float %a, float %b) #0 {
+; GFX803-LABEL: fptrunc_fmul_to_f16:
+; GFX803:       ; %bb.0: ; %.entry
+; GFX803-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX803-NEXT:    v_mul_f32_e32 v0, v0, v1
+; GFX803-NEXT:    v_cvt_f16_f32_e32 v0, v0
+; GFX803-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX900-LABEL: fptrunc_fmul_to_f16:
+; GFX900:       ; %bb.0: ; %.entry
+; GFX900-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX900-NEXT:    v_mad_mixlo_f16 v0, v0, v1, neg(0)
+; GFX900-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX906-LABEL: fptrunc_fmul_to_f16:
+; GFX906:       ; %bb.0: ; %.entry
+; GFX906-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX906-NEXT:    v_fma_mixlo_f16 v0, v0, v1, neg(0)
+; GFX906-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX90A-LABEL: fptrunc_fmul_to_f16:
+; GFX90A:       ; %bb.0: ; %.entry
+; GFX90A-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX90A-NEXT:    v_fma_mixlo_f16 v0, v0, v1, neg(0)
+; GFX90A-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1100-LABEL: fptrunc_fmul_to_f16:
+; GFX1100:       ; %bb.0: ; %.entry
+; GFX1100-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX1100-NEXT:    v_fma_mixlo_f16 v0, v0, v1, neg(0)
+; GFX1100-NEXT:    s_setpc_b64 s[30:31]
+;
+; GISEL-GFX900-LABEL: fptrunc_fmul_to_f16:
+; GISEL-GFX900:       ; %bb.0: ; %.entry
+; GISEL-GFX900-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GISEL-GFX900-NEXT:    v_mad_mixlo_f16 v0, v0, v1, neg(0)
+; GISEL-GFX900-NEXT:    s_setpc_b64 s[30:31]
+;
+; GISEL-GFX906-LABEL: fptrunc_fmul_to_f16:
+; GISEL-GFX906:       ; %bb.0: ; %.entry
+; GISEL-GFX906-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GISEL-GFX906-NEXT:    v_fma_mixlo_f16 v0, v0, v1, neg(0)
+; GISEL-GFX906-NEXT:    s_setpc_b64 s[30:31]
+.entry:
+  %mul = fmul float %a, %b
+  %cvt = fptrunc float %mul to half
+  ret half %cvt
+}
+
+define half @fptrunc_fmul_to_f16_contract(float %a, float %b) #0 {
+; GFX803-LABEL: fptrunc_fmul_to_f16_contract:
+; GFX803:       ; %bb.0: ; %.entry
+; GFX803-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX803-NEXT:    v_mul_f32_e32 v0, v0, v1
+; GFX803-NEXT:    v_cvt_f16_f32_e32 v0, v0
+; GFX803-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX900-LABEL: fptrunc_fmul_to_f16_contract:
+; GFX900:       ; %bb.0: ; %.entry
+; GFX900-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX900-NEXT:    v_mad_mixlo_f16 v0, v0, v1, neg(0)
+; GFX900-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX906-LABEL: fptrunc_fmul_to_f16_contract:
+; GFX906:       ; %bb.0: ; %.entry
+; GFX906-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX906-NEXT:    v_fma_mixlo_f16 v0, v0, v1, neg(0)
+; GFX906-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX90A-LABEL: fptrunc_fmul_to_f16_contract:
+; GFX90A:       ; %bb.0: ; %.entry
+; GFX90A-NEXT:    s_waitcnt vmcnt(0) expcnt(0) lgkmcnt(0)
+; GFX90A-NEXT:    v_fma_mixlo_f16 v0, v0, v1, neg(0)
+; GFX90A-NEXT:    s_setpc_b64 s[30:31]
+;
+; GFX1100-LABEL: fptru...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/219723


More information about the llvm-commits mailing list