[llvm] 2305da8 - [NFC][SLP][AMDGPU] Precommit a cross-block fmul fadd contraction test (#226609)

via llvm-commits llvm-commits at lists.llvm.org
Sat Sep 26 07:00:27 PDT 2026


Author: Dmitry Sidorov
Date: 2026-09-26T14:00:18Z
New Revision: 2305da864cb3b66e1fe857c3018bf7d94253bb8a

URL: https://github.com/llvm/llvm-project/commit/2305da864cb3b66e1fe857c3018bf7d94253bb8a
DIFF: https://github.com/llvm/llvm-project/commit/2305da864cb3b66e1fe857c3018bf7d94253bb8a.diff

LOG: [NFC][SLP][AMDGPU] Precommit a cross-block fmul fadd contraction test (#226609)

Two contract fmuls feed contract fadds in different successors. AMDGPU
sinks such an fmul into the block of its user and fuses the pair, so the
scalar form is one fma per path. Record the current behaviour, the pair
is kept scalar when the fmul is operand 0 of the fadd and paired when it
is operand 1.

Added: 
    llvm/test/Transforms/SLPVectorizer/AMDGPU/cross-block-fmul-fadd.ll

Modified: 
    

Removed: 
    


################################################################################
diff  --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/cross-block-fmul-fadd.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/cross-block-fmul-fadd.ll
new file mode 100644
index 0000000000000..8af404e1ce3c8
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/cross-block-fmul-fadd.ll
@@ -0,0 +1,124 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgpu9.0a-amd-amdhsa < %s | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgpu9.42-amd-amdhsa < %s | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgpu12.50-amd-amdhsa < %s | FileCheck %s
+
+; Two products are computed up front and each one feeds a contractable fadd in
+; a 
diff erent successor. The backend sinks such a single use fmul into the
+; block of its fadd and fuses the pair, and it merges the adjacent scalar
+; loads on its own, so the scalar code is one fma per path. Pairing the
+; multiplies keeps the loads and trades the fma of each path for a packed
+; multiply before the branch plus an add on the path.
+
+define void @cross_block_fmul_lhs(ptr addrspace(1) %p, ptr addrspace(1) %q, ptr addrspace(1) %r, ptr addrspace(1) %r2, float %x, float %y, i1 %c) {
+; CHECK-LABEL: define void @cross_block_fmul_lhs(
+; CHECK-SAME: ptr addrspace(1) [[P:%.*]], ptr addrspace(1) [[Q:%.*]], ptr addrspace(1) [[R:%.*]], ptr addrspace(1) [[R2:%.*]], float [[X:%.*]], float [[Y:%.*]], i1 [[C:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TID:%.*]] = call i32 @llvm.amdgcn.workitem.id.x()
+; CHECK-NEXT:    [[I0:%.*]] = shl i32 [[TID]], 1
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[P]], i32 [[I0]]
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[P0]], i32 1
+; CHECK-NEXT:    [[Q0:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[Q]], i32 [[I0]]
+; CHECK-NEXT:    [[Q1:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[Q0]], i32 1
+; CHECK-NEXT:    [[A0:%.*]] = load float, ptr addrspace(1) [[P0]], align 4
+; CHECK-NEXT:    [[A1:%.*]] = load float, ptr addrspace(1) [[P1]], align 4
+; CHECK-NEXT:    [[B0:%.*]] = load float, ptr addrspace(1) [[Q0]], align 4
+; CHECK-NEXT:    [[B1:%.*]] = load float, ptr addrspace(1) [[Q1]], align 4
+; CHECK-NEXT:    [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT:    [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT:    br i1 [[C]], label %[[T:.*]], label %[[F:.*]]
+; CHECK:       [[T]]:
+; CHECK-NEXT:    [[S0:%.*]] = fadd contract float [[M0]], [[X]]
+; CHECK-NEXT:    [[RT:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[R]], i32 [[TID]]
+; CHECK-NEXT:    store float [[S0]], ptr addrspace(1) [[RT]], align 4
+; CHECK-NEXT:    ret void
+; CHECK:       [[F]]:
+; CHECK-NEXT:    [[S1:%.*]] = fadd contract float [[M1]], [[Y]]
+; CHECK-NEXT:    [[RF:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[R2]], i32 [[TID]]
+; CHECK-NEXT:    store float [[S1]], ptr addrspace(1) [[RF]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %tid = call i32 @llvm.amdgcn.workitem.id.x()
+  %i0 = shl i32 %tid, 1
+  %p0 = getelementptr inbounds float, ptr addrspace(1) %p, i32 %i0
+  %p1 = getelementptr inbounds float, ptr addrspace(1) %p0, i32 1
+  %q0 = getelementptr inbounds float, ptr addrspace(1) %q, i32 %i0
+  %q1 = getelementptr inbounds float, ptr addrspace(1) %q0, i32 1
+  %a0 = load float, ptr addrspace(1) %p0, align 4
+  %a1 = load float, ptr addrspace(1) %p1, align 4
+  %b0 = load float, ptr addrspace(1) %q0, align 4
+  %b1 = load float, ptr addrspace(1) %q1, align 4
+  %m0 = fmul contract float %a0, %b0
+  %m1 = fmul contract float %a1, %b1
+  br i1 %c, label %t, label %f
+
+t:
+  %s0 = fadd contract float %m0, %x
+  %rt = getelementptr inbounds float, ptr addrspace(1) %r, i32 %tid
+  store float %s0, ptr addrspace(1) %rt, align 4
+  ret void
+
+f:
+  %s1 = fadd contract float %m1, %y
+  %rf = getelementptr inbounds float, ptr addrspace(1) %r2, i32 %tid
+  store float %s1, ptr addrspace(1) %rf, align 4
+  ret void
+}
+
+; The same with the fmul in operand 1 of the fadd.
+
+define void @cross_block_fmul_rhs(ptr addrspace(1) %p, ptr addrspace(1) %q, ptr addrspace(1) %r, ptr addrspace(1) %r2, float %x, float %y, i1 %c) {
+; CHECK-LABEL: define void @cross_block_fmul_rhs(
+; CHECK-SAME: ptr addrspace(1) [[P:%.*]], ptr addrspace(1) [[Q:%.*]], ptr addrspace(1) [[R:%.*]], ptr addrspace(1) [[R2:%.*]], float [[X:%.*]], float [[Y:%.*]], i1 [[C:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TID:%.*]] = call i32 @llvm.amdgcn.workitem.id.x()
+; CHECK-NEXT:    [[I0:%.*]] = shl i32 [[TID]], 1
+; CHECK-NEXT:    [[P0:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[P]], i32 [[I0]]
+; CHECK-NEXT:    [[Q0:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[Q]], i32 [[I0]]
+; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x float>, ptr addrspace(1) [[P0]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x float>, ptr addrspace(1) [[Q0]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = fmul contract <2 x float> [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    br i1 [[C]], label %[[T:.*]], label %[[F:.*]]
+; CHECK:       [[T]]:
+; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <2 x float> [[TMP2]], i64 0
+; CHECK-NEXT:    [[S0:%.*]] = fadd contract float [[X]], [[TMP3]]
+; CHECK-NEXT:    [[RT:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[R]], i32 [[TID]]
+; CHECK-NEXT:    store float [[S0]], ptr addrspace(1) [[RT]], align 4
+; CHECK-NEXT:    ret void
+; CHECK:       [[F]]:
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <2 x float> [[TMP2]], i64 1
+; CHECK-NEXT:    [[S1:%.*]] = fadd contract float [[Y]], [[TMP4]]
+; CHECK-NEXT:    [[RF:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[R2]], i32 [[TID]]
+; CHECK-NEXT:    store float [[S1]], ptr addrspace(1) [[RF]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %tid = call i32 @llvm.amdgcn.workitem.id.x()
+  %i0 = shl i32 %tid, 1
+  %p0 = getelementptr inbounds float, ptr addrspace(1) %p, i32 %i0
+  %p1 = getelementptr inbounds float, ptr addrspace(1) %p0, i32 1
+  %q0 = getelementptr inbounds float, ptr addrspace(1) %q, i32 %i0
+  %q1 = getelementptr inbounds float, ptr addrspace(1) %q0, i32 1
+  %a0 = load float, ptr addrspace(1) %p0, align 4
+  %a1 = load float, ptr addrspace(1) %p1, align 4
+  %b0 = load float, ptr addrspace(1) %q0, align 4
+  %b1 = load float, ptr addrspace(1) %q1, align 4
+  %m0 = fmul contract float %a0, %b0
+  %m1 = fmul contract float %a1, %b1
+  br i1 %c, label %t, label %f
+
+t:
+  %s0 = fadd contract float %x, %m0
+  %rt = getelementptr inbounds float, ptr addrspace(1) %r, i32 %tid
+  store float %s0, ptr addrspace(1) %rt, align 4
+  ret void
+
+f:
+  %s1 = fadd contract float %y, %m1
+  %rf = getelementptr inbounds float, ptr addrspace(1) %r2, i32 %tid
+  store float %s1, ptr addrspace(1) %rf, align 4
+  ret void
+}
+
+declare i32 @llvm.amdgcn.workitem.id.x()


        


More information about the llvm-commits mailing list