[llvm] [NFC][SLP] Precommit test for fmul operand position in the FMA check (PR #218407)

Dmitry Sidorov via llvm-commits llvm-commits at lists.llvm.org
Mon Aug 24 06:24:59 PDT 2026


https://github.com/MrSidims created https://github.com/llvm/llvm-project/pull/218407

SLP keeps a one-use fmul scalar so the backend can fuse it, but only operand 0 of the fadd/fsub is checked. An fmul on the right hand side is gathered into a vector fmul instead and the contraction is lost.

Test only, records current behavior. Fixed in a follow-up.

>From d6304e41e81ea3e2e768ed7e8fbf0ee029c49058 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Mon, 24 Aug 2026 08:22:12 -0500
Subject: [PATCH] [NFC][SLP] Precommit test for fmul operand position in the
 FMA check

SLP keeps a one-use fmul scalar so the backend can fuse it, but only
operand 0 of the fadd/fsub is checked. An fmul on the right hand side is
gathered into a vector fmul instead and the contraction is lost.

Test only, records current behavior. Fixed in a follow-up.
---
 .../SLPVectorizer/X86/fma-operand-index.ll    | 157 ++++++++++++++++++
 1 file changed, 157 insertions(+)
 create mode 100644 llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll

diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
new file mode 100644
index 0000000000000..abf8abf106eee
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
@@ -0,0 +1,157 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu -mcpu=znver4 < %s | FileCheck %s
+
+; A one-use fmul is left scalar so the backend can fuse it, but only operand 0
+; of the fadd/fsub is checked. The same fmul is kept or gathered depending on
+; which side it sits on. To be addressed in a follow-up.
+
+; Both fmuls are operand 1, so they gather and cannot fuse.
+define double @fmul_rhs_fadd(ptr %x, ptr %y, ptr %z) {
+; CHECK-LABEL: define double @fmul_rhs_fadd(
+; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
+; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
+; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
+; CHECK-NEXT:    [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
+; CHECK-NEXT:    [[MUL0:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
+; CHECK-NEXT:    [[ADD0:%.*]] = fadd reassoc nsz contract double [[ZSUM]], [[MUL0]]
+; CHECK-NEXT:    [[MUL1:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
+; CHECK-NEXT:    [[ADD1:%.*]] = fadd reassoc nsz contract double [[ADD0]], [[MUL1]]
+; CHECK-NEXT:    ret double [[ADD1]]
+;
+entry:
+  %x8 = getelementptr inbounds nuw i8, ptr %x, i64 8
+  %y8 = getelementptr inbounds nuw i8, ptr %y, i64 8
+  %z8 = getelementptr inbounds nuw i8, ptr %z, i64 8
+  %x0 = load double, ptr %x, align 8
+  %y0 = load double, ptr %y, align 8
+  %mul0 = fmul reassoc nsz contract double %y0, %x0
+  %z0 = load double, ptr %z, align 8
+  %x1 = load double, ptr %x8, align 8
+  %y1 = load double, ptr %y8, align 8
+  %mul1 = fmul reassoc nsz contract double %y1, %x1
+  %z1 = load double, ptr %z8, align 8
+  %zsum = fadd reassoc nsz contract double %z0, %z1
+  %add0 = fadd reassoc nsz contract double %zsum, %mul0
+  %add1 = fadd reassoc nsz contract double %add0, %mul1
+  ret double %add1
+}
+
+; Same on the subtrahend of an fsub, which would fuse to an fnmadd.
+define double @fmul_rhs_fsub(ptr %x, ptr %y, ptr %z) {
+; CHECK-LABEL: define double @fmul_rhs_fsub(
+; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
+; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
+; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
+; CHECK-NEXT:    [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
+; CHECK-NEXT:    [[MUL0:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
+; CHECK-NEXT:    [[SUB0:%.*]] = fsub reassoc nsz contract double [[ZSUM]], [[MUL0]]
+; CHECK-NEXT:    [[MUL1:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
+; CHECK-NEXT:    [[SUB1:%.*]] = fsub reassoc nsz contract double [[SUB0]], [[MUL1]]
+; CHECK-NEXT:    ret double [[SUB1]]
+;
+entry:
+  %x8 = getelementptr inbounds nuw i8, ptr %x, i64 8
+  %y8 = getelementptr inbounds nuw i8, ptr %y, i64 8
+  %z8 = getelementptr inbounds nuw i8, ptr %z, i64 8
+  %x0 = load double, ptr %x, align 8
+  %y0 = load double, ptr %y, align 8
+  %mul0 = fmul reassoc nsz contract double %y0, %x0
+  %z0 = load double, ptr %z, align 8
+  %x1 = load double, ptr %x8, align 8
+  %y1 = load double, ptr %y8, align 8
+  %mul1 = fmul reassoc nsz contract double %y1, %x1
+  %z1 = load double, ptr %z8, align 8
+  %zsum = fadd reassoc nsz contract double %z0, %z1
+  %sub0 = fsub reassoc nsz contract double %zsum, %mul0
+  %sub1 = fsub reassoc nsz contract double %sub0, %mul1
+  ret double %sub1
+}
+
+; Operand 0 is the one checked, so these stay scalar and fuse.
+define double @fmul_lhs_fadd(ptr %x, ptr %y, ptr %z) {
+; CHECK-LABEL: define double @fmul_lhs_fadd(
+; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT:    [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
+; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT:    [[X0:%.*]] = load double, ptr [[X]], align 8
+; CHECK-NEXT:    [[Y0:%.*]] = load double, ptr [[Y]], align 8
+; CHECK-NEXT:    [[MUL0:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
+; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
+; CHECK-NEXT:    [[X1:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT:    [[Y1:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT:    [[MUL1:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
+; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
+; CHECK-NEXT:    [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
+; CHECK-NEXT:    [[ADD0:%.*]] = fadd reassoc nsz contract double [[MUL0]], [[ZSUM]]
+; CHECK-NEXT:    [[ADD1:%.*]] = fadd reassoc nsz contract double [[MUL1]], [[ADD0]]
+; CHECK-NEXT:    ret double [[ADD1]]
+;
+entry:
+  %x8 = getelementptr inbounds nuw i8, ptr %x, i64 8
+  %y8 = getelementptr inbounds nuw i8, ptr %y, i64 8
+  %z8 = getelementptr inbounds nuw i8, ptr %z, i64 8
+  %x0 = load double, ptr %x, align 8
+  %y0 = load double, ptr %y, align 8
+  %mul0 = fmul reassoc nsz contract double %y0, %x0
+  %z0 = load double, ptr %z, align 8
+  %x1 = load double, ptr %x8, align 8
+  %y1 = load double, ptr %y8, align 8
+  %mul1 = fmul reassoc nsz contract double %y1, %x1
+  %z1 = load double, ptr %z8, align 8
+  %zsum = fadd reassoc nsz contract double %z0, %z1
+  %add0 = fadd reassoc nsz contract double %mul0, %zsum
+  %add1 = fadd reassoc nsz contract double %mul1, %add0
+  ret double %add1
+}
+
+; A second use blocks fusion anyway, so gathering is right here.
+define double @fmul_rhs_fadd_multi_use(ptr %x, ptr %y, ptr %z, ptr %out) {
+; CHECK-LABEL: define double @fmul_rhs_fadd_multi_use(
+; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]], ptr [[OUT:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
+; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
+; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[X]], align 8
+; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[Y]], align 8
+; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    store <2 x double> [[TMP2]], ptr [[OUT]], align 8
+; CHECK-NEXT:    [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
+; CHECK-NEXT:    [[ADD0:%.*]] = fadd reassoc nsz contract double [[ZSUM]], [[TMP3]]
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
+; CHECK-NEXT:    [[ADD1:%.*]] = fadd reassoc nsz contract double [[ADD0]], [[TMP4]]
+; CHECK-NEXT:    ret double [[ADD1]]
+;
+entry:
+  %x8 = getelementptr inbounds nuw i8, ptr %x, i64 8
+  %y8 = getelementptr inbounds nuw i8, ptr %y, i64 8
+  %z8 = getelementptr inbounds nuw i8, ptr %z, i64 8
+  %o8 = getelementptr inbounds nuw i8, ptr %out, i64 8
+  %x0 = load double, ptr %x, align 8
+  %y0 = load double, ptr %y, align 8
+  %mul0 = fmul reassoc nsz contract double %y0, %x0
+  %z0 = load double, ptr %z, align 8
+  %x1 = load double, ptr %x8, align 8
+  %y1 = load double, ptr %y8, align 8
+  %mul1 = fmul reassoc nsz contract double %y1, %x1
+  %z1 = load double, ptr %z8, align 8
+  store double %mul0, ptr %out, align 8
+  store double %mul1, ptr %o8, align 8
+  %zsum = fadd reassoc nsz contract double %z0, %z1
+  %add0 = fadd reassoc nsz contract double %zsum, %mul0
+  %add1 = fadd reassoc nsz contract double %add0, %mul1
+  ret double %add1
+}



More information about the llvm-commits mailing list