[llvm] [AMDGPU] Add gfx1310 coverage to the bf16 transcendental tests (PR #223054)

via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 11 13:47:09 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-amdgpu

Author: Joe Nash (Sisyph)

<details>
<summary>Changes</summary>

57a9c4f939e5 made codegen emit the VOP3 encoding with op_sel[0] set when a bf16 inline constant reaches one of the single-source VOP1 bf16 opcodes, because the hardware generates the constant in the high half of the corresponding fp32 inline constant.

GFX1310 has the same FeatureBF16InlineConstFromUpperFP32 behaviour. Mirror the enabled gfx1250 SelectionDAG RUN lines with gfx1310 ones so the workaround is covered there too. Test-only change, no functional change.

Assisted-by: Claude Code:claude-opus-5[1m]

---

Patch is 30.69 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/223054.diff


6 Files Affected:

- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cos.bf16.ll (+76) 
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.bf16.ll (+57) 
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.log.bf16.ll (+57) 
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.rcp.bf16.ll (+95) 
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.sin.bf16.ll (+74) 
- (modified) llvm/test/CodeGen/AMDGPU/llvm.amdgcn.sqrt.bf16.ll (+57) 


``````````diff
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cos.bf16.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cos.bf16.ll
index b45e8109c9e2e..149b4153be671 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cos.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.cos.bf16.ll
@@ -3,6 +3,10 @@
 ; xUN: llc -global-isel=1 -mtriple=amdgpu12.50 -mattr=-real-true16 < %s | FileCheck -check-prefixes=GCN,FAKE16 %s
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.50 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GCN,REAL16 %s
 ; xUN: llc -global-isel=1 -mtriple=amdgpu12.50 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GCN,REAL16 %s
+; RUN: llc -global-isel=0 -mtriple=amdgpu13.10 -mattr=-real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-FAKE16 %s
+; xUN: llc -global-isel=1 -mtriple=amdgpu13.10 -mattr=-real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-FAKE16 %s
+; RUN: llc -global-isel=0 -mtriple=amdgpu13.10 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-REAL16 %s
+; xUN: llc -global-isel=1 -mtriple=amdgpu13.10 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-REAL16 %s
 
 ; FIXME: GlobalISel does not work with bf16
 
@@ -34,6 +38,24 @@ define amdgpu_kernel void @cos_bf16(ptr addrspace(1) %out, bfloat %src) #1 {
 ; REAL16-NEXT:    v_cos_bf16_e32 v0.l, s2
 ; REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; REAL16-NEXT:    s_endpgm
+;
+; GFX13-FAKE16-LABEL: cos_bf16:
+; GFX13-FAKE16:       ; %bb.0:
+; GFX13-FAKE16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
+; GFX13-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-FAKE16-NEXT:    v_cos_bf16_e32 v0, s2
+; GFX13-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-REAL16-LABEL: cos_bf16:
+; GFX13-REAL16:       ; %bb.0:
+; GFX13-REAL16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
+; GFX13-REAL16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-REAL16-NEXT:    v_cos_bf16_e32 v0.l, s2
+; GFX13-REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-REAL16-NEXT:    s_endpgm
   %cos = call bfloat @llvm.amdgcn.cos.bf16(bfloat %src) #0
   store bfloat %cos, ptr addrspace(1) %out, align 2
   ret void
@@ -65,6 +87,24 @@ define amdgpu_kernel void @cos_bf16_constant_4_strictfp(ptr addrspace(1) %out) #
 ; REAL16-NEXT:    s_wait_kmcnt 0x0
 ; REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; REAL16-NEXT:    s_endpgm
+;
+; GFX13-FAKE16-LABEL: cos_bf16_constant_4_strictfp:
+; GFX13-FAKE16:       ; %bb.0:
+; GFX13-FAKE16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-FAKE16-NEXT:    v_cos_bf16_e64 v0, 4.0 op_sel:[1,0]
+; GFX13-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-REAL16-LABEL: cos_bf16_constant_4_strictfp:
+; GFX13-REAL16:       ; %bb.0:
+; GFX13-REAL16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-REAL16-NEXT:    v_cos_bf16_e64 v0.l, 4.0 op_sel:[1,0]
+; GFX13-REAL16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-REAL16-NEXT:    s_endpgm
   %cos = call bfloat @llvm.amdgcn.cos.bf16(bfloat 4.0) strictfp
   store bfloat %cos, ptr addrspace(1) %out, align 2
   ret void
@@ -96,6 +136,24 @@ define amdgpu_kernel void @cos_bf16_constant_100_strictfp(ptr addrspace(1) %out)
 ; REAL16-NEXT:    s_wait_kmcnt 0x0
 ; REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; REAL16-NEXT:    s_endpgm
+;
+; GFX13-FAKE16-LABEL: cos_bf16_constant_100_strictfp:
+; GFX13-FAKE16:       ; %bb.0:
+; GFX13-FAKE16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-FAKE16-NEXT:    v_cos_bf16_e32 v0, 0x42c8
+; GFX13-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-REAL16-LABEL: cos_bf16_constant_100_strictfp:
+; GFX13-REAL16:       ; %bb.0:
+; GFX13-REAL16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-REAL16-NEXT:    v_cos_bf16_e32 v0.l, 0x42c8
+; GFX13-REAL16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-REAL16-NEXT:    s_endpgm
   %cos = call bfloat @llvm.amdgcn.cos.bf16(bfloat 100.0) strictfp
   store bfloat %cos, ptr addrspace(1) %out, align 2
   ret void
@@ -126,6 +184,23 @@ define amdgpu_kernel void @cos_bf16_constant_0.3(ptr addrspace(1) %out) #1 {
 ; REAL16-NEXT:    s_wait_kmcnt 0x0
 ; REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; REAL16-NEXT:    s_endpgm
+;
+; GFX13-FAKE16-LABEL: cos_bf16_constant_0.3:
+; GFX13-FAKE16:       ; %bb.0:
+; GFX13-FAKE16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-FAKE16-NEXT:    v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, 0xffffbea1
+; GFX13-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-FAKE16-NEXT:    global_store_b16 v0, v1, s[0:1]
+; GFX13-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-REAL16-LABEL: cos_bf16_constant_0.3:
+; GFX13-REAL16:       ; %bb.0:
+; GFX13-REAL16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-REAL16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-REAL16-NEXT:    v_mov_b16_e32 v0.l, 0xbea1
+; GFX13-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-REAL16-NEXT:    s_endpgm
   %cos = call bfloat @llvm.amdgcn.cos.bf16(bfloat 0.3) #0
   store bfloat %cos, ptr addrspace(1) %out, align 2
   ret void
@@ -136,3 +211,4 @@ attributes #1 = { nounwind }
 attributes #2 = { nounwind strictfp }
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; GCN: {{.*}}
+; GFX13: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.bf16.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.bf16.ll
index 591569a107a9a..f84e4dc1e62d6 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.exp.bf16.ll
@@ -3,6 +3,8 @@
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.50 -mattr=-real-true16 < %s | FileCheck -check-prefixes=GCN,FAKE16 %s
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.50 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GCN,REAL16 %s
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.50 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GCN,REAL16 %s
+; RUN: llc -global-isel=0 -mtriple=amdgpu13.10 -mattr=-real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-FAKE16 %s
+; RUN: llc -global-isel=0 -mtriple=amdgpu13.10 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-REAL16 %s
 
 declare bfloat @llvm.amdgcn.exp2.bf16(bfloat) #0
 
@@ -32,6 +34,24 @@ define amdgpu_kernel void @exp_bf16(ptr addrspace(1) %out, bfloat %src) #1 {
 ; REAL16-NEXT:    v_exp_bf16_e32 v0.l, s2
 ; REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; REAL16-NEXT:    s_endpgm
+;
+; GFX13-FAKE16-LABEL: exp_bf16:
+; GFX13-FAKE16:       ; %bb.0:
+; GFX13-FAKE16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
+; GFX13-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-FAKE16-NEXT:    v_exp_bf16_e32 v0, s2
+; GFX13-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-REAL16-LABEL: exp_bf16:
+; GFX13-REAL16:       ; %bb.0:
+; GFX13-REAL16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
+; GFX13-REAL16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-REAL16-NEXT:    v_exp_bf16_e32 v0.l, s2
+; GFX13-REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-REAL16-NEXT:    s_endpgm
   %exp = call bfloat @llvm.amdgcn.exp2.bf16(bfloat %src) #0
   store bfloat %exp, ptr addrspace(1) %out, align 2
   ret void
@@ -63,6 +83,24 @@ define amdgpu_kernel void @exp_bf16_constant_4(ptr addrspace(1) %out) #1 {
 ; REAL16-NEXT:    s_wait_kmcnt 0x0
 ; REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; REAL16-NEXT:    s_endpgm
+;
+; GFX13-FAKE16-LABEL: exp_bf16_constant_4:
+; GFX13-FAKE16:       ; %bb.0:
+; GFX13-FAKE16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-FAKE16-NEXT:    v_exp_bf16_e64 v0, 4.0 op_sel:[1,0]
+; GFX13-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-REAL16-LABEL: exp_bf16_constant_4:
+; GFX13-REAL16:       ; %bb.0:
+; GFX13-REAL16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-REAL16-NEXT:    v_exp_bf16_e64 v0.l, 4.0 op_sel:[1,0]
+; GFX13-REAL16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-REAL16-NEXT:    s_endpgm
   %exp = call bfloat @llvm.amdgcn.exp2.bf16(bfloat 4.0) #0
   store bfloat %exp, ptr addrspace(1) %out, align 2
   ret void
@@ -94,6 +132,24 @@ define amdgpu_kernel void @exp_bf16_constant_100(ptr addrspace(1) %out) #1 {
 ; REAL16-NEXT:    s_wait_kmcnt 0x0
 ; REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; REAL16-NEXT:    s_endpgm
+;
+; GFX13-FAKE16-LABEL: exp_bf16_constant_100:
+; GFX13-FAKE16:       ; %bb.0:
+; GFX13-FAKE16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-FAKE16-NEXT:    v_exp_bf16_e32 v0, 0x42c8
+; GFX13-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-REAL16-LABEL: exp_bf16_constant_100:
+; GFX13-REAL16:       ; %bb.0:
+; GFX13-REAL16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-REAL16-NEXT:    v_exp_bf16_e32 v0.l, 0x42c8
+; GFX13-REAL16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-REAL16-NEXT:    s_endpgm
   %exp = call bfloat @llvm.amdgcn.exp2.bf16(bfloat 100.0) #0
   store bfloat %exp, ptr addrspace(1) %out, align 2
   ret void
@@ -103,3 +159,4 @@ attributes #0 = { nounwind readnone }
 attributes #1 = { nounwind }
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; GCN: {{.*}}
+; GFX13: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.log.bf16.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.log.bf16.ll
index ac3d14df41758..1ed9202e0b674 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.log.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.log.bf16.ll
@@ -3,6 +3,8 @@
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.50 -mattr=-real-true16 < %s | FileCheck -check-prefixes=GCN,FAKE16 %s
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.50 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GCN,REAL16 %s
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.50 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GCN,REAL16 %s
+; RUN: llc -global-isel=0 -mtriple=amdgpu13.10 -mattr=-real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-FAKE16 %s
+; RUN: llc -global-isel=0 -mtriple=amdgpu13.10 -mattr=+real-true16 < %s | FileCheck -check-prefixes=GFX13,GFX13-REAL16 %s
 
 declare bfloat @llvm.amdgcn.log.bf16(bfloat) #0
 
@@ -32,6 +34,24 @@ define amdgpu_kernel void @log_bf16(ptr addrspace(1) %out, bfloat %src) #1 {
 ; REAL16-NEXT:    v_log_bf16_e32 v0.l, s2
 ; REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; REAL16-NEXT:    s_endpgm
+;
+; GFX13-FAKE16-LABEL: log_bf16:
+; GFX13-FAKE16:       ; %bb.0:
+; GFX13-FAKE16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
+; GFX13-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-FAKE16-NEXT:    v_log_bf16_e32 v0, s2
+; GFX13-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-REAL16-LABEL: log_bf16:
+; GFX13-REAL16:       ; %bb.0:
+; GFX13-REAL16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x24 nv
+; GFX13-REAL16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-REAL16-NEXT:    v_log_bf16_e32 v0.l, s2
+; GFX13-REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-REAL16-NEXT:    s_endpgm
   %log = call bfloat @llvm.amdgcn.log.bf16(bfloat %src) #0
   store bfloat %log, ptr addrspace(1) %out, align 2
   ret void
@@ -63,6 +83,24 @@ define amdgpu_kernel void @log_bf16_constant_4(ptr addrspace(1) %out) #1 {
 ; REAL16-NEXT:    s_wait_kmcnt 0x0
 ; REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; REAL16-NEXT:    s_endpgm
+;
+; GFX13-FAKE16-LABEL: log_bf16_constant_4:
+; GFX13-FAKE16:       ; %bb.0:
+; GFX13-FAKE16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-FAKE16-NEXT:    v_log_bf16_e64 v0, 4.0 op_sel:[1,0]
+; GFX13-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-REAL16-LABEL: log_bf16_constant_4:
+; GFX13-REAL16:       ; %bb.0:
+; GFX13-REAL16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-REAL16-NEXT:    v_log_bf16_e64 v0.l, 4.0 op_sel:[1,0]
+; GFX13-REAL16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-REAL16-NEXT:    s_endpgm
   %log = call bfloat @llvm.amdgcn.log.bf16(bfloat 4.0) #0
   store bfloat %log, ptr addrspace(1) %out, align 2
   ret void
@@ -94,6 +132,24 @@ define amdgpu_kernel void @log_bf16_constant_100(ptr addrspace(1) %out) #1 {
 ; REAL16-NEXT:    s_wait_kmcnt 0x0
 ; REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; REAL16-NEXT:    s_endpgm
+;
+; GFX13-FAKE16-LABEL: log_bf16_constant_100:
+; GFX13-FAKE16:       ; %bb.0:
+; GFX13-FAKE16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-FAKE16-NEXT:    v_log_bf16_e32 v0, 0x42c8
+; GFX13-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-REAL16-LABEL: log_bf16_constant_100:
+; GFX13-REAL16:       ; %bb.0:
+; GFX13-REAL16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x24 nv
+; GFX13-REAL16-NEXT:    v_log_bf16_e32 v0.l, 0x42c8
+; GFX13-REAL16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-REAL16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-REAL16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-REAL16-NEXT:    s_endpgm
   %log = call bfloat @llvm.amdgcn.log.bf16(bfloat 100.0) #0
   store bfloat %log, ptr addrspace(1) %out, align 2
   ret void
@@ -103,3 +159,4 @@ attributes #0 = { nounwind readnone }
 attributes #1 = { nounwind }
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
 ; GCN: {{.*}}
+; GFX13: {{.*}}
diff --git a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.rcp.bf16.ll b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.rcp.bf16.ll
index 25088830a3a33..f6e68ab3f4ac1 100644
--- a/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.rcp.bf16.ll
+++ b/llvm/test/CodeGen/AMDGPU/llvm.amdgcn.rcp.bf16.ll
@@ -3,6 +3,8 @@
 ; RUN: llc -global-isel=0 -mtriple=amdgpu12.50-amd-amdhsa -mattr=-real-true16 < %s | FileCheck -check-prefix=SDAG-FAKE16 %s
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.50-amd-amdhsa -mattr=+real-true16 < %s | FileCheck -check-prefix=GI-TRUE16 %s
 ; RUN: llc -global-isel=1 -mtriple=amdgpu12.50-amd-amdhsa -mattr=-real-true16 < %s | FileCheck -check-prefix=GI-FAKE16 %s
+; RUN: llc -global-isel=0 -mtriple=amdgpu13.10-amd-amdhsa -mattr=+real-true16 < %s | FileCheck -check-prefix=GFX13-SDAG-TRUE16 %s
+; RUN: llc -global-isel=0 -mtriple=amdgpu13.10-amd-amdhsa -mattr=-real-true16 < %s | FileCheck -check-prefix=GFX13-SDAG-FAKE16 %s
 
 declare bfloat @llvm.amdgcn.rcp.bf16(bfloat) #0
 
@@ -58,6 +60,24 @@ define amdgpu_kernel void @rcp_bf16(ptr addrspace(1) %out, bfloat %src) #1 {
 ; GI-FAKE16-NEXT:    v_rcp_bf16_e32 v0, s2
 ; GI-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; GI-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-SDAG-TRUE16-LABEL: rcp_bf16:
+; GFX13-SDAG-TRUE16:       ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x0 nv
+; GFX13-SDAG-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-SDAG-TRUE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT:    v_rcp_bf16_e32 v0.l, s2
+; GFX13-SDAG-TRUE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-SDAG-TRUE16-NEXT:    s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: rcp_bf16:
+; GFX13-SDAG-FAKE16:       ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT:    s_load_b96 s[0:2], s[4:5], 0x0 nv
+; GFX13-SDAG-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-SDAG-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT:    v_rcp_bf16_e32 v0, s2
+; GFX13-SDAG-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT:    s_endpgm
   %rcp = call bfloat @llvm.amdgcn.rcp.bf16(bfloat %src) #0
   store bfloat %rcp, ptr addrspace(1) %out, align 2
   ret void
@@ -127,6 +147,30 @@ define amdgpu_kernel void @rcp_bf16_global_load(ptr addrspace(1) %out, ptr addrs
 ; GI-FAKE16-NEXT:    v_rcp_bf16_e32 v0, v0
 ; GI-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; GI-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-SDAG-TRUE16-LABEL: rcp_bf16_global_load:
+; GFX13-SDAG-TRUE16:       ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX13-SDAG-TRUE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-SDAG-TRUE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT:    global_load_d16_b16 v0, v0, s[2:3] scale_offset
+; GFX13-SDAG-TRUE16-NEXT:    s_wait_loadcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT:    v_rcp_bf16_e32 v0.l, v0.l
+; GFX13-SDAG-TRUE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-SDAG-TRUE16-NEXT:    s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: rcp_bf16_global_load:
+; GFX13-SDAG-FAKE16:       ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT:    s_load_b128 s[0:3], s[4:5], 0x0 nv
+; GFX13-SDAG-FAKE16-NEXT:    v_and_b32_e32 v0, 0x3ff, v0
+; GFX13-SDAG-FAKE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-SDAG-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT:    global_load_u16 v0, v0, s[2:3] scale_offset
+; GFX13-SDAG-FAKE16-NEXT:    s_wait_loadcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT:    v_rcp_bf16_e32 v0, v0
+; GFX13-SDAG-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT:    s_endpgm
   %tid = call i32 @llvm.amdgcn.workitem.id.x()
   %src.ptr = getelementptr bfloat, ptr addrspace(1) %in, i32 %tid
   %src = load bfloat, ptr addrspace(1) %src.ptr, align 2
@@ -186,6 +230,23 @@ define amdgpu_kernel void @rcp_bf16_constant_4(ptr addrspace(1) %out) #1 {
 ; GI-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GI-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; GI-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-SDAG-TRUE16-LABEL: rcp_bf16_constant_4:
+; GFX13-SDAG-TRUE16:       ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
+; GFX13-SDAG-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-SDAG-TRUE16-NEXT:    v_mov_b16_e32 v0.l, 0x3e80
+; GFX13-SDAG-TRUE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-SDAG-TRUE16-NEXT:    s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: rcp_bf16_constant_4:
+; GFX13-SDAG-FAKE16:       ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
+; GFX13-SDAG-FAKE16-NEXT:    v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, 0x3e80
+; GFX13-SDAG-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT:    global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT:    s_endpgm
   %rcp = call bfloat @llvm.amdgcn.rcp.bf16(bfloat 4.0) #0
   store bfloat %rcp, ptr addrspace(1) %out, align 2
   ret void
@@ -242,6 +303,23 @@ define amdgpu_kernel void @rcp_bf16_constant_100(ptr addrspace(1) %out) #1 {
 ; GI-FAKE16-NEXT:    s_wait_kmcnt 0x0
 ; GI-FAKE16-NEXT:    global_store_b16 v1, v0, s[0:1]
 ; GI-FAKE16-NEXT:    s_endpgm
+;
+; GFX13-SDAG-TRUE16-LABEL: rcp_bf16_constant_100:
+; GFX13-SDAG-TRUE16:       ; %bb.0:
+; GFX13-SDAG-TRUE16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
+; GFX13-SDAG-TRUE16-NEXT:    v_mov_b32_e32 v1, 0
+; GFX13-SDAG-TRUE16-NEXT:    v_mov_b16_e32 v0.l, 0x3c24
+; GFX13-SDAG-TRUE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-SDAG-TRUE16-NEXT:    global_store_b16 v1, v0, s[0:1]
+; GFX13-SDAG-TRUE16-NEXT:    s_endpgm
+;
+; GFX13-SDAG-FAKE16-LABEL: rcp_bf16_constant_100:
+; GFX13-SDAG-FAKE16:       ; %bb.0:
+; GFX13-SDAG-FAKE16-NEXT:    s_load_b64 s[0:1], s[4:5], 0x0 nv
+; GFX13-SDAG-FAKE16-NEXT:    v_dual_mov_b32 v0, 0 :: v_dual_mov_b32 v1, 0x3c24
+; GFX13-SDAG-FAKE16-NEXT:    s_wait_kmcnt 0x0
+; GFX13-SDAG-FAKE16-NEXT:    global_store_b16 v0, v1, s[0:1]
+; GFX13-SDAG-FAKE16-NEXT:    s_endpgm
   %rcp = call bfloat @llvm.amdgcn.rcp.bf16(bfloat 100.0) #0
   store bfloat %rcp, ptr addrspace(1) %out, align 2
   ret void
@@ -298,6 +376,23 @@ define amdgpu_kernel void @r...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/223054


More information about the llvm-commits mailing list