[llvm] [AMDGPU] Keep divergent i64 mul feeding an add for the mad64 fold (PR #226963)

Pankaj Dwivedi via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 29 05:10:35 PDT 2026


================
@@ -0,0 +1,123 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 6
+; RUN: llc -mtriple=amdgpu9.08 < %s | FileCheck -check-prefix=GCN %s
+
+; On GFX9+ AMDGPUCodeGenPrepare keeps a divergent i64 mul whose only user is an
+; add as a plain mul, so the add combine fuses it into v_mad_u64_u32 or
+; v_mad_i64_i32 instead of a mul24 + carry chain.
+
+declare i32 @llvm.amdgcn.workitem.id.x()
+
+; Unsigned 24-bit product feeding an add.
+define amdgpu_kernel void @mul24_u_add_to_mad(ptr addrspace(1) %p, ptr addrspace(1) %q) {
+; GCN-LABEL: mul24_u_add_to_mad:
+; GCN:       ; %bb.0:
+; GCN-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GCN-NEXT:    v_lshlrev_b32_e32 v2, 3, v0
+; GCN-NEXT:    s_waitcnt lgkmcnt(0)
+; GCN-NEXT:    global_load_dword v3, v2, s[0:1] offset:4
+; GCN-NEXT:    global_load_dwordx2 v[0:1], v2, s[2:3]
+; GCN-NEXT:    s_movk_i32 s2, 0xd1
+; GCN-NEXT:    s_waitcnt vmcnt(1)
+; GCN-NEXT:    v_lshrrev_b32_e32 v3, 8, v3
+; GCN-NEXT:    s_waitcnt vmcnt(0)
+; GCN-NEXT:    v_mad_u64_u32 v[0:1], s[2:3], v3, s2, v[0:1]
+; GCN-NEXT:    global_store_dwordx2 v2, v[0:1], s[0:1]
+; GCN-NEXT:    s_endpgm
+  %tid = call i32 @llvm.amdgcn.workitem.id.x()
+  %pg = getelementptr i64, ptr addrspace(1) %p, i32 %tid
+  %qg = getelementptr i64, ptr addrspace(1) %q, i32 %tid
+  %x = load i64, ptr addrspace(1) %pg
+  %acc = load i64, ptr addrspace(1) %qg
+  %hi = lshr i64 %x, 40                  ; <= 24 significant bits (u24)
+  %mul = mul i64 %hi, 209                ; 209 fits in 24 bits
+  %add = add i64 %mul, %acc
+  store i64 %add, ptr addrspace(1) %pg
+  ret void
+}
+
+; Signed 24-bit product feeding an add.
+define amdgpu_kernel void @mul24_i_add_to_mad(ptr addrspace(1) %p, ptr addrspace(1) %q) {
+; GCN-LABEL: mul24_i_add_to_mad:
+; GCN:       ; %bb.0:
+; GCN-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GCN-NEXT:    v_lshlrev_b32_e32 v2, 3, v0
+; GCN-NEXT:    s_waitcnt lgkmcnt(0)
+; GCN-NEXT:    global_load_dword v3, v2, s[0:1] offset:4
+; GCN-NEXT:    global_load_dwordx2 v[0:1], v2, s[2:3]
+; GCN-NEXT:    s_movk_i32 s2, 0x64
+; GCN-NEXT:    s_waitcnt vmcnt(1)
+; GCN-NEXT:    v_ashrrev_i32_e32 v3, 8, v3
+; GCN-NEXT:    s_waitcnt vmcnt(0)
+; GCN-NEXT:    v_mad_i64_i32 v[0:1], s[2:3], v3, s2, v[0:1]
+; GCN-NEXT:    global_store_dwordx2 v2, v[0:1], s[0:1]
+; GCN-NEXT:    s_endpgm
+  %tid = call i32 @llvm.amdgcn.workitem.id.x()
+  %pg = getelementptr i64, ptr addrspace(1) %p, i32 %tid
+  %qg = getelementptr i64, ptr addrspace(1) %q, i32 %tid
+  %x = load i64, ptr addrspace(1) %pg
+  %acc = load i64, ptr addrspace(1) %qg
+  %sh = ashr i64 %x, 40                  ; <= 24 significant signed bits (i24)
+  %mul = mul i64 %sh, 100                ; 100 fits in signed 24 bits
+  %add = add i64 %mul, %acc
+  store i64 %add, ptr addrspace(1) %pg
+  ret void
+}
+
+; A GEP user keeps the mul24 form. The ISel ptradd pattern forms the mad.
+define amdgpu_kernel void @mul24_u_ptradd_to_mad(ptr addrspace(1) %p) {
+; GCN-LABEL: mul24_u_ptradd_to_mad:
+; GCN:       ; %bb.0:
+; GCN-NEXT:    s_load_dwordx2 s[0:1], s[4:5], 0x24
+; GCN-NEXT:    v_lshlrev_b32_e32 v0, 3, v0
+; GCN-NEXT:    s_movk_i32 s2, 0xd1
+; GCN-NEXT:    s_waitcnt lgkmcnt(0)
+; GCN-NEXT:    global_load_dwordx2 v[0:1], v0, s[0:1]
+; GCN-NEXT:    v_mov_b32_e32 v3, s1
+; GCN-NEXT:    v_mov_b32_e32 v2, s0
+; GCN-NEXT:    s_waitcnt vmcnt(0)
+; GCN-NEXT:    v_lshrrev_b32_e32 v4, 8, v1
+; GCN-NEXT:    v_mad_u64_u32 v[2:3], s[0:1], v4, s2, v[2:3]
+; GCN-NEXT:    global_store_dwordx2 v[2:3], v[0:1], off
+; GCN-NEXT:    s_endpgm
+  %tid = call i32 @llvm.amdgcn.workitem.id.x()
+  %pg = getelementptr i64, ptr addrspace(1) %p, i32 %tid
+  %x = load i64, ptr addrspace(1) %pg
+  %hi = lshr i64 %x, 40
+  %off = mul i64 %hi, 209
+  %gep = getelementptr i8, ptr addrspace(1) %p, i64 %off
+  store i64 %x, ptr addrspace(1) %gep
+  ret void
+}
+
+; A second user keeps the mul24 form. The ISel pattern fuses the add and the
+; product is recomputed for the store.
+define amdgpu_kernel void @mul24_u_add_extra_use(ptr addrspace(1) %p, ptr addrspace(1) %q) {
+; GCN-LABEL: mul24_u_add_extra_use:
+; GCN:       ; %bb.0:
+; GCN-NEXT:    s_load_dwordx4 s[0:3], s[4:5], 0x24
+; GCN-NEXT:    v_lshlrev_b32_e32 v4, 3, v0
+; GCN-NEXT:    s_movk_i32 s4, 0xd1
+; GCN-NEXT:    s_waitcnt lgkmcnt(0)
+; GCN-NEXT:    global_load_dword v2, v4, s[0:1] offset:4
+; GCN-NEXT:    global_load_dwordx2 v[0:1], v4, s[2:3]
+; GCN-NEXT:    s_waitcnt vmcnt(1)
+; GCN-NEXT:    v_lshrrev_b32_e32 v2, 8, v2
+; GCN-NEXT:    s_waitcnt vmcnt(0)
+; GCN-NEXT:    v_mad_u64_u32 v[0:1], s[4:5], v2, s4, v[0:1]
+; GCN-NEXT:    v_mul_hi_u32_u24_e32 v3, 0xd1, v2
+; GCN-NEXT:    v_mul_u32_u24_e32 v2, 0xd1, v2
----------------
PankajDwivedi-25 wrote:

This still expects v_mad_u64_u32 plus a rematerialized mul24/mulhi24 pair for the store.

https://github.com/llvm/llvm-project/pull/226963


More information about the llvm-commits mailing list