[llvm] [SLP][NFC]Add extra FMA candidates tests, NFC (PR #226684)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Sat Sep 26 05:26:34 PDT 2026


https://github.com/alexey-bataev created https://github.com/llvm/llvm-project/pull/226684

None

>From 30b47fadf8a8c779e5ce0e139e73ab1dce2b0cde Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sat, 26 Sep 2026 05:26:21 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
 =?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../reduced-value-next-to-fused-fmul.ll       | 42 +++++++++++++++
 .../AMDGPU/fmul-sunk-into-fadd-block.ll       | 52 +++++++++++++++++++
 2 files changed, 94 insertions(+)
 create mode 100644 llvm/test/Transforms/SLPVectorizer/AArch64/reduced-value-next-to-fused-fmul.ll
 create mode 100644 llvm/test/Transforms/SLPVectorizer/AMDGPU/fmul-sunk-into-fadd-block.ll

diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/reduced-value-next-to-fused-fmul.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/reduced-value-next-to-fused-fmul.ll
new file mode 100644
index 0000000000000..288364e64958b
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/reduced-value-next-to-fused-fmul.ll
@@ -0,0 +1,42 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -mtriple=aarch64-unknown-linux -mcpu=neoverse-v2 -S < %s | FileCheck %s
+
+; The first reduction operation fuses with its fmul operand, not with the
+; reduced fdiv, so the cost of the fdiv is not excluded from the scalar
+; reduction as the cost of the fused fmul.
+define double @fdivs_next_to_fused_fmul(ptr %x, ptr %y, double %a0, double %b0) {
+; CHECK-LABEL: define double @fdivs_next_to_fused_fmul(
+; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], double [[A0:%.*]], double [[B0:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:    [[P0:%.*]] = fmul reassoc contract double [[A0]], [[B0]]
+; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[X]], align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x double>, ptr [[Y]], align 8
+; CHECK-NEXT:    [[TMP3:%.*]] = fdiv reassoc contract <4 x double> [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[TMP4:%.*]] = call reassoc contract double @llvm.vector.reduce.fadd.v4f64(double -0.000000e+00, <4 x double> [[TMP3]])
+; CHECK-NEXT:    [[OP_RDX:%.*]] = fadd reassoc contract double [[TMP4]], [[P0]]
+; CHECK-NEXT:    ret double [[OP_RDX]]
+;
+  %p0 = fmul reassoc contract double %a0, %b0
+  %x1 = load double, ptr %x
+  %y1 = load double, ptr %y
+  %q1 = fdiv reassoc contract double %x1, %y1
+  %x2p = getelementptr double, ptr %x, i64 1
+  %x2 = load double, ptr %x2p
+  %y2p = getelementptr double, ptr %y, i64 1
+  %y2 = load double, ptr %y2p
+  %q2 = fdiv reassoc contract double %x2, %y2
+  %x3p = getelementptr double, ptr %x, i64 2
+  %x3 = load double, ptr %x3p
+  %y3p = getelementptr double, ptr %y, i64 2
+  %y3 = load double, ptr %y3p
+  %q3 = fdiv reassoc contract double %x3, %y3
+  %x4p = getelementptr double, ptr %x, i64 3
+  %x4 = load double, ptr %x4p
+  %y4p = getelementptr double, ptr %y, i64 3
+  %y4 = load double, ptr %y4p
+  %q4 = fdiv reassoc contract double %x4, %y4
+  %r0 = fadd reassoc contract double %p0, %q1
+  %r1 = fadd reassoc contract double %r0, %q2
+  %r2 = fadd reassoc contract double %r1, %q3
+  %r3 = fadd reassoc contract double %r2, %q4
+  ret double %r3
+}
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/fmul-sunk-into-fadd-block.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/fmul-sunk-into-fadd-block.ll
new file mode 100644
index 0000000000000..63597fb54b840
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/fmul-sunk-into-fadd-block.ll
@@ -0,0 +1,52 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S -passes=slp-vectorizer -mtriple=amdgpu9.0a-amd-amdhsa < %s | FileCheck %s
+
+; The fmul operands of the fadds are computed in different blocks, so they
+; cannot form a vector node and are gathered. The target sinks each fmul into
+; the block of its fadd user, where they fuse into an fma in the scalar code.
+; The vectorized fadds keep the gathered fmuls unfused, so the fadds stay
+; scalar.
+define void @fmuls_sunk_into_fadd_block(ptr addrspace(1) %out, ptr addrspace(1) %c, float %a0, float %b0, float %a1, float %b1, i1 %cond) {
+; CHECK-LABEL: define void @fmuls_sunk_into_fadd_block(
+; CHECK-SAME: ptr addrspace(1) [[OUT:%.*]], ptr addrspace(1) [[C:%.*]], float [[A0:%.*]], float [[B0:%.*]], float [[A1:%.*]], float [[B1:%.*]], i1 [[COND:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT:    br i1 [[COND]], label %[[MUL:.*]], label %[[EXIT:.*]]
+; CHECK:       [[MUL]]:
+; CHECK-NEXT:    [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT:    br label %[[ADD:.*]]
+; CHECK:       [[ADD]]:
+; CHECK-NEXT:    [[C0:%.*]] = load float, ptr addrspace(1) [[C]], align 8
+; CHECK-NEXT:    [[C_1:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[C]], i64 1
+; CHECK-NEXT:    [[C1:%.*]] = load float, ptr addrspace(1) [[C_1]], align 4
+; CHECK-NEXT:    [[S0:%.*]] = fadd contract float [[M0]], [[C0]]
+; CHECK-NEXT:    [[S1:%.*]] = fadd contract float [[M1]], [[C1]]
+; CHECK-NEXT:    store float [[S0]], ptr addrspace(1) [[OUT]], align 8
+; CHECK-NEXT:    [[OUT_1:%.*]] = getelementptr inbounds float, ptr addrspace(1) [[OUT]], i64 1
+; CHECK-NEXT:    store float [[S1]], ptr addrspace(1) [[OUT_1]], align 4
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %m0 = fmul contract float %a0, %b0
+  br i1 %cond, label %mul, label %exit
+
+mul:
+  %m1 = fmul contract float %a1, %b1
+  br label %add
+
+add:
+  %c0 = load float, ptr addrspace(1) %c, align 8
+  %c.1 = getelementptr inbounds float, ptr addrspace(1) %c, i64 1
+  %c1 = load float, ptr addrspace(1) %c.1, align 4
+  %s0 = fadd contract float %m0, %c0
+  %s1 = fadd contract float %m1, %c1
+  store float %s0, ptr addrspace(1) %out, align 8
+  %out.1 = getelementptr inbounds float, ptr addrspace(1) %out, i64 1
+  store float %s1, ptr addrspace(1) %out.1, align 4
+  br label %exit
+
+exit:
+  ret void
+}



More information about the llvm-commits mailing list