[llvm] [SLP] Fix canConvertToFMA operand selection and fmul costing (PR #216425)

Dmitry Sidorov via llvm-commits llvm-commits at lists.llvm.org
Mon Aug 17 02:59:29 PDT 2026


https://github.com/MrSidims updated https://github.com/llvm/llvm-project/pull/216425

>From 554fc8e30174ad10479dfcff6859da0b4f2f83b0 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Fri, 14 Aug 2026 12:51:56 +0200
Subject: [PATCH 1/7] [SLP] Fix canConvertToFMA operand selection and fmul
 costing

canConvertToFMA only looked for the fmul on operand 0 of the fadd, so the
accumulator shape fadd acc, a * b was never recognized. Check both operands
for chains that do not allow reassociation.

Also price the unfused fmul without a context instruction. Targets that
model fma fusion price a contractable fmul as free, which discounted the
scalar side of the comparison too and fmuladd never looked profitable.
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    |  49 +++++--
 .../AMDGPU/elementwise-fma-operand1.ll        | 130 ++++++++++++++++++
 2 files changed, 169 insertions(+), 10 deletions(-)
 create mode 100644 llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 53816d49de722..6e5628104f0f9 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14382,18 +14382,44 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
   InstructionsCompatibilityAnalysis Analysis(DT, DL, TTI, TLI);
   SmallVector<BoUpSLP::ValueList> Operands = Analysis.buildOperands(S, VL);
 
-  InstructionsState OpS = getSameOpcode(Operands.front(), TLI);
-  if (!OpS.valid())
-    return InstructionCost::getInvalid();
-
-  if (OpS.isAltShuffle() || OpS.getOpcode() != Instruction::FMul)
-    return InstructionCost::getInvalid();
-  if (!CheckForContractable(Operands.front()))
+  // The fmul may sit on either side of the add/sub. Look past operand 0 only
+  // for chains that do not allow reassociation. A reassociative chain can be
+  // vectorized into a vector fmul feeding a reduction, which is usually
+  // better than the scalar fma chain this check protects.
+  bool AllowReassoc = any_of(VL, [](Value *V) {
+    auto *FPCI = dyn_cast<FPMathOperator>(V);
+    return FPCI && FPCI->getFastMathFlags().allowReassoc();
+  });
+  auto GetFMulOperandIdx = [&]() -> std::optional<unsigned> {
+    for (unsigned Idx : seq<unsigned>(0, AllowReassoc ? 1 : Operands.size())) {
+      InstructionsState CandS = getSameOpcode(Operands[Idx], TLI);
+      if (!CandS.valid() || CandS.isAltShuffle() ||
+          CandS.getOpcode() != Instruction::FMul)
+        continue;
+      if (!CheckForContractable(Operands[Idx]))
+        continue;
+      return Idx;
+    }
+    return std::nullopt;
+  };
+  std::optional<unsigned> FMulIdx = GetFMulOperandIdx();
+  if (!FMulIdx)
     return InstructionCost::getInvalid();
+  InstructionsState OpS = getSameOpcode(Operands[*FMulIdx], TLI);
   // Compare the costs.
   InstructionCost FMulPlusFAddCost = 0;
   InstructionCost FMACost = 0;
   constexpr TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
+  // Price the fmul as not fused with its user. Passing a context instruction
+  // would let targets that model the fusion discount the unfused side of the
+  // comparison as well.
+  auto GetUnfusedFMulCost = [&](Instruction *I) {
+    TTI::OperandValueInfo Op1Info = TTI::getOperandInfo(I->getOperand(0));
+    TTI::OperandValueInfo Op2Info = TTI::getOperandInfo(I->getOperand(1));
+    return TTI.getArithmeticInstrCost(Instruction::FMul, I->getType(), CostKind,
+                                      Op1Info, Op2Info,
+                                      {I->getOperand(0), I->getOperand(1)});
+  };
   FastMathFlags FMF;
   FMF.set();
   for (Value *V : VL) {
@@ -14406,7 +14432,7 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
     FMulPlusFAddCost += TTI.getInstructionCost(I, CostKind);
   }
   unsigned NumOps = 0;
-  for (auto [V, Op] : zip(VL, Operands.front())) {
+  for (auto [V, Op] : zip(VL, Operands[*FMulIdx])) {
     if (S.isCopyableElement(V))
       continue;
     auto *I = dyn_cast<Instruction>(Op);
@@ -14420,7 +14446,9 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
     ++NumOps;
     if (auto *FPCI = dyn_cast<FPMathOperator>(I))
       FMF &= FPCI->getFastMathFlags();
-    FMulPlusFAddCost += TTI.getInstructionCost(I, CostKind);
+    FMulPlusFAddCost += I->getOpcode() == Instruction::FMul
+                            ? GetUnfusedFMulCost(I)
+                            : TTI.getInstructionCost(I, CostKind);
   }
   Type *Ty = VL.front()->getType();
   IntrinsicCostAttributes ICA(Intrinsic::fmuladd, Ty, {Ty, Ty, Ty}, FMF);
@@ -15225,7 +15253,8 @@ void BoUpSLP::transformNodes() {
         break;
       // This node is a fmuladd node.
       E.CombinedOp = TreeEntry::FMulAdd;
-      TreeEntry *FMulEntry = getOperandEntry(&E, 0);
+      TreeEntry *FMulEntry =
+          getOperandEntry(&E, IsOneUseVectorFMulOperand(LHS) ? 0 : 1);
       if (FMulEntry->UserTreeIndex &&
           FMulEntry->State == TreeEntry::Vectorize) {
         // The FMul node is part of the combined fmuladd node.
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
new file mode 100644
index 0000000000000..efa8f5c4dd930
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
@@ -0,0 +1,130 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a < %s | FileCheck %s
+
+; Elementwise d = c + a * b where the fmul is operand 1 of the fadd. Marking
+; a hardcoded operand 0 turned the c load bundle into a CombinedVectorize
+; node and made vectorizeTree abort.
+
+define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr noalias %c) {
+; CHECK-LABEL: define void @axpy4_contract(
+; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x float>, ptr [[C]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x float>, ptr [[A]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = load <2 x float>, ptr [[B]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = fmul contract <2 x float> [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[TMP4:%.*]] = fadd contract <2 x float> [[TMP0]], [[TMP3]]
+; CHECK-NEXT:    store <2 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT:    [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT:    [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT:    [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT:    [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT:    [[TMP5:%.*]] = load <2 x float>, ptr [[CP2]], align 4
+; CHECK-NEXT:    [[TMP6:%.*]] = load <2 x float>, ptr [[AP2]], align 4
+; CHECK-NEXT:    [[TMP7:%.*]] = load <2 x float>, ptr [[BP2]], align 4
+; CHECK-NEXT:    [[TMP8:%.*]] = fmul contract <2 x float> [[TMP6]], [[TMP7]]
+; CHECK-NEXT:    [[TMP9:%.*]] = fadd contract <2 x float> [[TMP5]], [[TMP8]]
+; CHECK-NEXT:    store <2 x float> [[TMP9]], ptr [[DP2]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %c0 = load float, ptr %c, align 4
+  %a0 = load float, ptr %a, align 4
+  %b0 = load float, ptr %b, align 4
+  %m0 = fmul contract float %a0, %b0
+  %r0 = fadd contract float %c0, %m0
+  store float %r0, ptr %d, align 4
+  %cp1 = getelementptr inbounds float, ptr %c, i64 1
+  %c1 = load float, ptr %cp1, align 4
+  %ap1 = getelementptr inbounds float, ptr %a, i64 1
+  %a1 = load float, ptr %ap1, align 4
+  %bp1 = getelementptr inbounds float, ptr %b, i64 1
+  %b1 = load float, ptr %bp1, align 4
+  %m1 = fmul contract float %a1, %b1
+  %r1 = fadd contract float %c1, %m1
+  %dp1 = getelementptr inbounds float, ptr %d, i64 1
+  store float %r1, ptr %dp1, align 4
+  %cp2 = getelementptr inbounds float, ptr %c, i64 2
+  %c2 = load float, ptr %cp2, align 4
+  %ap2 = getelementptr inbounds float, ptr %a, i64 2
+  %a2 = load float, ptr %ap2, align 4
+  %bp2 = getelementptr inbounds float, ptr %b, i64 2
+  %b2 = load float, ptr %bp2, align 4
+  %m2 = fmul contract float %a2, %b2
+  %r2 = fadd contract float %c2, %m2
+  %dp2 = getelementptr inbounds float, ptr %d, i64 2
+  store float %r2, ptr %dp2, align 4
+  %cp3 = getelementptr inbounds float, ptr %c, i64 3
+  %c3 = load float, ptr %cp3, align 4
+  %ap3 = getelementptr inbounds float, ptr %a, i64 3
+  %a3 = load float, ptr %ap3, align 4
+  %bp3 = getelementptr inbounds float, ptr %b, i64 3
+  %b3 = load float, ptr %bp3, align 4
+  %m3 = fmul contract float %a3, %b3
+  %r3 = fadd contract float %c3, %m3
+  %dp3 = getelementptr inbounds float, ptr %d, i64 3
+  store float %r3, ptr %dp3, align 4
+  ret void
+}
+
+define void @axpy4_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr noalias %c) {
+; CHECK-LABEL: define void @axpy4_reassoc(
+; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x float>, ptr [[C]], align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x float>, ptr [[A]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = load <2 x float>, ptr [[B]], align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = fmul reassoc contract <2 x float> [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[TMP4:%.*]] = fadd reassoc contract <2 x float> [[TMP0]], [[TMP3]]
+; CHECK-NEXT:    store <2 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT:    [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT:    [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT:    [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT:    [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT:    [[TMP5:%.*]] = load <2 x float>, ptr [[CP2]], align 4
+; CHECK-NEXT:    [[TMP6:%.*]] = load <2 x float>, ptr [[AP2]], align 4
+; CHECK-NEXT:    [[TMP7:%.*]] = load <2 x float>, ptr [[BP2]], align 4
+; CHECK-NEXT:    [[TMP8:%.*]] = fmul reassoc contract <2 x float> [[TMP6]], [[TMP7]]
+; CHECK-NEXT:    [[TMP9:%.*]] = fadd reassoc contract <2 x float> [[TMP5]], [[TMP8]]
+; CHECK-NEXT:    store <2 x float> [[TMP9]], ptr [[DP2]], align 4
+; CHECK-NEXT:    ret void
+;
+entry:
+  %c0 = load float, ptr %c, align 4
+  %a0 = load float, ptr %a, align 4
+  %b0 = load float, ptr %b, align 4
+  %m0 = fmul contract reassoc float %a0, %b0
+  %r0 = fadd contract reassoc float %c0, %m0
+  store float %r0, ptr %d, align 4
+  %cp1 = getelementptr inbounds float, ptr %c, i64 1
+  %c1 = load float, ptr %cp1, align 4
+  %ap1 = getelementptr inbounds float, ptr %a, i64 1
+  %a1 = load float, ptr %ap1, align 4
+  %bp1 = getelementptr inbounds float, ptr %b, i64 1
+  %b1 = load float, ptr %bp1, align 4
+  %m1 = fmul contract reassoc float %a1, %b1
+  %r1 = fadd contract reassoc float %c1, %m1
+  %dp1 = getelementptr inbounds float, ptr %d, i64 1
+  store float %r1, ptr %dp1, align 4
+  %cp2 = getelementptr inbounds float, ptr %c, i64 2
+  %c2 = load float, ptr %cp2, align 4
+  %ap2 = getelementptr inbounds float, ptr %a, i64 2
+  %a2 = load float, ptr %ap2, align 4
+  %bp2 = getelementptr inbounds float, ptr %b, i64 2
+  %b2 = load float, ptr %bp2, align 4
+  %m2 = fmul contract reassoc float %a2, %b2
+  %r2 = fadd contract reassoc float %c2, %m2
+  %dp2 = getelementptr inbounds float, ptr %d, i64 2
+  store float %r2, ptr %dp2, align 4
+  %cp3 = getelementptr inbounds float, ptr %c, i64 3
+  %c3 = load float, ptr %cp3, align 4
+  %ap3 = getelementptr inbounds float, ptr %a, i64 3
+  %a3 = load float, ptr %ap3, align 4
+  %bp3 = getelementptr inbounds float, ptr %b, i64 3
+  %b3 = load float, ptr %bp3, align 4
+  %m3 = fmul contract reassoc float %a3, %b3
+  %r3 = fadd contract reassoc float %c3, %m3
+  %dp3 = getelementptr inbounds float, ptr %d, i64 3
+  store float %r3, ptr %dp3, align 4
+  ret void
+}

>From 6f88c8a36a58695280f6390846e80bbbbf580c13 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sat, 15 Aug 2026 16:48:53 +0200
Subject: [PATCH 2/7] apply comments

---
 llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 11 ++++-------
 1 file changed, 4 insertions(+), 7 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 6e5628104f0f9..67c52d4e8cc52 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14386,10 +14386,8 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
   // for chains that do not allow reassociation. A reassociative chain can be
   // vectorized into a vector fmul feeding a reduction, which is usually
   // better than the scalar fma chain this check protects.
-  bool AllowReassoc = any_of(VL, [](Value *V) {
-    auto *FPCI = dyn_cast<FPMathOperator>(V);
-    return FPCI && FPCI->getFastMathFlags().allowReassoc();
-  });
+  bool AllowReassoc = any_of(
+      VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
   auto GetFMulOperandIdx = [&]() -> std::optional<unsigned> {
     for (unsigned Idx : seq<unsigned>(0, AllowReassoc ? 1 : Operands.size())) {
       InstructionsState CandS = getSameOpcode(Operands[Idx], TLI);
@@ -14414,6 +14412,7 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
   // would let targets that model the fusion discount the unfused side of the
   // comparison as well.
   auto GetUnfusedFMulCost = [&](Instruction *I) {
+    assert(I->getOpcode() == Instruction::FMul && "Expected an fmul");
     TTI::OperandValueInfo Op1Info = TTI::getOperandInfo(I->getOperand(0));
     TTI::OperandValueInfo Op2Info = TTI::getOperandInfo(I->getOperand(1));
     return TTI.getArithmeticInstrCost(Instruction::FMul, I->getType(), CostKind,
@@ -14446,9 +14445,7 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
     ++NumOps;
     if (auto *FPCI = dyn_cast<FPMathOperator>(I))
       FMF &= FPCI->getFastMathFlags();
-    FMulPlusFAddCost += I->getOpcode() == Instruction::FMul
-                            ? GetUnfusedFMulCost(I)
-                            : TTI.getInstructionCost(I, CostKind);
+    FMulPlusFAddCost += GetUnfusedFMulCost(I);
   }
   Type *Ty = VL.front()->getType();
   IntrinsicCostAttributes ICA(Intrinsic::fmuladd, Ty, {Ty, Ty, Ty}, FMF);

>From 7f9561465cc4f8f72b5723d2fa201b4300ba245b Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sat, 15 Aug 2026 23:28:39 +0200
Subject: [PATCH 3/7] extra fixes

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 15 +++++---
 .../AMDGPU/elementwise-fma-operand1.ll        | 36 ++++++++++++++++---
 2 files changed, 41 insertions(+), 10 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 67c52d4e8cc52..b74dd7f80f5b8 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14383,13 +14383,17 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
   SmallVector<BoUpSLP::ValueList> Operands = Analysis.buildOperands(S, VL);
 
   // The fmul may sit on either side of the add/sub. Look past operand 0 only
-  // for chains that do not allow reassociation. A reassociative chain can be
-  // vectorized into a vector fmul feeding a reduction, which is usually
-  // better than the scalar fma chain this check protects.
+  // for chains that do not allow reassociation. The gate is profitability, not
+  // correctness. A reassociative chain can be vectorized into a vector fmul
+  // feeding a reduction, which is usually better than the scalar fma chain
+  // this check protects. An fsub can only fold an fmul on its left, so stop at
+  // operand 0 there as well.
   bool AllowReassoc = any_of(
       VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
+  bool OnlyFirstOperand = AllowReassoc || S.getOpcode() == Instruction::FSub;
   auto GetFMulOperandIdx = [&]() -> std::optional<unsigned> {
-    for (unsigned Idx : seq<unsigned>(0, AllowReassoc ? 1 : Operands.size())) {
+    for (unsigned Idx :
+         seq<unsigned>(0, OnlyFirstOperand ? 1 : Operands.size())) {
       InstructionsState CandS = getSameOpcode(Operands[Idx], TLI);
       if (!CandS.valid() || CandS.isAltShuffle() ||
           CandS.getOpcode() != Instruction::FMul)
@@ -14428,7 +14432,8 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
     if (!S.isCopyableElement(I))
       if (auto *FPCI = dyn_cast<FPMathOperator>(I))
         FMF &= FPCI->getFastMathFlags();
-    FMulPlusFAddCost += TTI.getInstructionCost(I, CostKind);
+    FMulPlusFAddCost +=
+        TTI.getArithmeticInstrCost(S.getOpcode(), I->getType(), CostKind);
   }
   unsigned NumOps = 0;
   for (auto [V, Op] : zip(VL, Operands[*FMulIdx])) {
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
index 44f55e536179a..b43c0a8fdf91c 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
@@ -2,12 +2,16 @@
 ; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -slp-threshold=14 < %s | FileCheck %s
 ; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -slp-threshold=14 < %s | FileCheck %s
 ; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -slp-threshold=14 < %s | FileCheck %s
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -slp-threshold=12 < %s | FileCheck %s --check-prefix=THR12
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx942 -slp-threshold=12 < %s | FileCheck %s --check-prefix=THR12
+; RUN: opt -passes=slp-vectorizer -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -slp-threshold=12 < %s | FileCheck %s --check-prefix=THR12
 
-; Elementwise d = c + a * b, where the fmul is operand 1 of the fadd. The
-; threshold puts the decision right at the cost boundary, so how the fmul and
-; fadd are priced against a fused fma is what decides it. These targets halve
-; the cost of a packed fmul, which is what tempts SLP into vectorizing and
-; breaking the scalar fma chain.
+; Elementwise d = c + a * b, where the fmul is operand 1 of the fadd. These
+; targets halve the cost of a packed fmul, so SLP is tempted to vectorize and
+; break the scalar fma chain. The 14 runs sit at the cost boundary. The 12 runs
+; vectorize either way and guard against the fmuladd marking landing on the load
+; at operand 0 after the fma detection picked the fmul at operand 1, which
+; asserts.
 
 define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr noalias %c) {
 ; CHECK-LABEL: define void @axpy4_contract(
@@ -51,6 +55,17 @@ define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr
 ; CHECK-NEXT:    store float [[R3]], ptr [[DP3]], align 4
 ; CHECK-NEXT:    ret void
 ;
+; THR12-LABEL: define void @axpy4_contract(
+; THR12-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0:[0-9]+]] {
+; THR12-NEXT:  [[ENTRY:.*:]]
+; THR12-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
+; THR12-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
+; THR12-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
+; THR12-NEXT:    [[TMP3:%.*]] = fmul contract <4 x float> [[TMP1]], [[TMP2]]
+; THR12-NEXT:    [[TMP4:%.*]] = fadd contract <4 x float> [[TMP0]], [[TMP3]]
+; THR12-NEXT:    store <4 x float> [[TMP4]], ptr [[D]], align 4
+; THR12-NEXT:    ret void
+;
 entry:
   %c0 = load float, ptr %c, align 4
   %a0 = load float, ptr %a, align 4
@@ -103,6 +118,17 @@ define void @axpy4_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr n
 ; CHECK-NEXT:    store <4 x float> [[TMP4]], ptr [[D]], align 4
 ; CHECK-NEXT:    ret void
 ;
+; THR12-LABEL: define void @axpy4_reassoc(
+; THR12-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
+; THR12-NEXT:  [[ENTRY:.*:]]
+; THR12-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
+; THR12-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
+; THR12-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
+; THR12-NEXT:    [[TMP3:%.*]] = fmul reassoc contract <4 x float> [[TMP1]], [[TMP2]]
+; THR12-NEXT:    [[TMP4:%.*]] = fadd reassoc contract <4 x float> [[TMP0]], [[TMP3]]
+; THR12-NEXT:    store <4 x float> [[TMP4]], ptr [[D]], align 4
+; THR12-NEXT:    ret void
+;
 entry:
   %c0 = load float, ptr %c, align 4
   %a0 = load float, ptr %a, align 4

>From 325b41fcbad6bbc3a1f1d3880d9498fcf717890f Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 16 Aug 2026 11:36:35 +0200
Subject: [PATCH 4/7] format

---
 llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index b74dd7f80f5b8..fcf526bfecb75 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14388,8 +14388,8 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
   // feeding a reduction, which is usually better than the scalar fma chain
   // this check protects. An fsub can only fold an fmul on its left, so stop at
   // operand 0 there as well.
-  bool AllowReassoc = any_of(
-      VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
+  bool AllowReassoc =
+      any_of(VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
   bool OnlyFirstOperand = AllowReassoc || S.getOpcode() == Instruction::FSub;
   auto GetFMulOperandIdx = [&]() -> std::optional<unsigned> {
     for (unsigned Idx :

>From f57929be7204d307d599b151f439bd5b269096c7 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 16 Aug 2026 16:55:28 +0200
Subject: [PATCH 5/7] Require the whole bundle to be reassoc, and take the
 operand count from the helper

---
 llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 8 +++++---
 1 file changed, 5 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index fcf526bfecb75..0432a9deac6ea 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14389,11 +14389,13 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
   // this check protects. An fsub can only fold an fmul on its left, so stop at
   // operand 0 there as well.
   bool AllowReassoc =
-      any_of(VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
+      all_of(VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
   bool OnlyFirstOperand = AllowReassoc || S.getOpcode() == Instruction::FSub;
+  unsigned NumCandidateOps =
+      OnlyFirstOperand ? 1
+                       : getNumberOfPotentiallyCommutativeOps(S.getMainOp());
   auto GetFMulOperandIdx = [&]() -> std::optional<unsigned> {
-    for (unsigned Idx :
-         seq<unsigned>(0, OnlyFirstOperand ? 1 : Operands.size())) {
+    for (unsigned Idx : seq<unsigned>(0, NumCandidateOps)) {
       InstructionsState CandS = getSameOpcode(Operands[Idx], TLI);
       if (!CandS.valid() || CandS.isAltShuffle() ||
           CandS.getOpcode() != Instruction::FMul)

>From 7221528e02cf4a6322cec42297137c21aac54126 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 16 Aug 2026 18:22:10 +0200
Subject: [PATCH 6/7] Cover the mixed-reassoc bundle the all_of change fixes

---
 .../AMDGPU/elementwise-fma-operand1.ll        | 126 +++++++++++++++++-
 1 file changed, 125 insertions(+), 1 deletion(-)

diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
index b43c0a8fdf91c..eed8216c5b1d2 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
@@ -11,7 +11,8 @@
 ; break the scalar fma chain. The 14 runs sit at the cost boundary. The 12 runs
 ; vectorize either way and guard against the fmuladd marking landing on the load
 ; at operand 0 after the fma detection picked the fmul at operand 1, which
-; asserts.
+; asserts. axpy4_mixed_reassoc carries reassoc on one lane only, so the whole
+; bundle has to be reassociative before the search gives up on operand 1.
 
 define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr noalias %c) {
 ; CHECK-LABEL: define void @axpy4_contract(
@@ -168,3 +169,126 @@ entry:
   store float %r3, ptr %dp3, align 4
   ret void
 }
+
+define void @axpy4_mixed_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr noalias %c) {
+; CHECK-LABEL: define void @axpy4_mixed_reassoc(
+; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT:    [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT:    [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT:    [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT:    [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; CHECK-NEXT:    store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT:    [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT:    [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT:    [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT:    [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT:    [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT:    [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT:    [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT:    [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; CHECK-NEXT:    [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT:    store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT:    [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT:    [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT:    [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT:    [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT:    [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT:    [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT:    [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; CHECK-NEXT:    [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; CHECK-NEXT:    [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT:    store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT:    [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT:    [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT:    [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT:    [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT:    [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT:    [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT:    [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; CHECK-NEXT:    [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; CHECK-NEXT:    [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT:    store float [[R3]], ptr [[DP3]], align 4
+; CHECK-NEXT:    ret void
+;
+; THR12-LABEL: define void @axpy4_mixed_reassoc(
+; THR12-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
+; THR12-NEXT:  [[ENTRY:.*:]]
+; THR12-NEXT:    [[C0:%.*]] = load float, ptr [[C]], align 4
+; THR12-NEXT:    [[A0:%.*]] = load float, ptr [[A]], align 4
+; THR12-NEXT:    [[B0:%.*]] = load float, ptr [[B]], align 4
+; THR12-NEXT:    [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; THR12-NEXT:    [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; THR12-NEXT:    store float [[R0]], ptr [[D]], align 4
+; THR12-NEXT:    [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; THR12-NEXT:    [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; THR12-NEXT:    [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; THR12-NEXT:    [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; THR12-NEXT:    [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; THR12-NEXT:    [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; THR12-NEXT:    [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; THR12-NEXT:    [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; THR12-NEXT:    [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; THR12-NEXT:    store float [[R1]], ptr [[DP1]], align 4
+; THR12-NEXT:    [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; THR12-NEXT:    [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; THR12-NEXT:    [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; THR12-NEXT:    [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; THR12-NEXT:    [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; THR12-NEXT:    [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; THR12-NEXT:    [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; THR12-NEXT:    [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; THR12-NEXT:    [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; THR12-NEXT:    store float [[R2]], ptr [[DP2]], align 4
+; THR12-NEXT:    [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; THR12-NEXT:    [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; THR12-NEXT:    [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; THR12-NEXT:    [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; THR12-NEXT:    [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; THR12-NEXT:    [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; THR12-NEXT:    [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; THR12-NEXT:    [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; THR12-NEXT:    [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; THR12-NEXT:    store float [[R3]], ptr [[DP3]], align 4
+; THR12-NEXT:    ret void
+;
+entry:
+  %c0 = load float, ptr %c, align 4
+  %a0 = load float, ptr %a, align 4
+  %b0 = load float, ptr %b, align 4
+  %m0 = fmul contract float %a0, %b0
+  %r0 = fadd contract reassoc float %c0, %m0
+  store float %r0, ptr %d, align 4
+  %cp1 = getelementptr inbounds float, ptr %c, i64 1
+  %c1 = load float, ptr %cp1, align 4
+  %ap1 = getelementptr inbounds float, ptr %a, i64 1
+  %a1 = load float, ptr %ap1, align 4
+  %bp1 = getelementptr inbounds float, ptr %b, i64 1
+  %b1 = load float, ptr %bp1, align 4
+  %m1 = fmul contract float %a1, %b1
+  %r1 = fadd contract float %c1, %m1
+  %dp1 = getelementptr inbounds float, ptr %d, i64 1
+  store float %r1, ptr %dp1, align 4
+  %cp2 = getelementptr inbounds float, ptr %c, i64 2
+  %c2 = load float, ptr %cp2, align 4
+  %ap2 = getelementptr inbounds float, ptr %a, i64 2
+  %a2 = load float, ptr %ap2, align 4
+  %bp2 = getelementptr inbounds float, ptr %b, i64 2
+  %b2 = load float, ptr %bp2, align 4
+  %m2 = fmul contract float %a2, %b2
+  %r2 = fadd contract float %c2, %m2
+  %dp2 = getelementptr inbounds float, ptr %d, i64 2
+  store float %r2, ptr %dp2, align 4
+  %cp3 = getelementptr inbounds float, ptr %c, i64 3
+  %c3 = load float, ptr %cp3, align 4
+  %ap3 = getelementptr inbounds float, ptr %a, i64 3
+  %a3 = load float, ptr %ap3, align 4
+  %bp3 = getelementptr inbounds float, ptr %b, i64 3
+  %b3 = load float, ptr %bp3, align 4
+  %m3 = fmul contract float %a3, %b3
+  %r3 = fadd contract float %c3, %m3
+  %dp3 = getelementptr inbounds float, ptr %d, i64 3
+  store float %r3, ptr %dp3, align 4
+  ret void
+}

>From 8dfbc8435bfa00486938f82f3ce61df2b727df72 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Mon, 17 Aug 2026 01:17:12 +0200
Subject: [PATCH 7/7] Skip undef and copyable lanes in the reassoc check

---
 llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 8 ++++++--
 1 file changed, 6 insertions(+), 2 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index e575918bf582c..bedea0c5ad873 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14429,8 +14429,12 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
   // feeding a reduction, which is usually better than the scalar fma chain
   // this check protects. An fsub can only fold an fmul on its left, so stop at
   // operand 0 there as well.
-  bool AllowReassoc =
-      all_of(VL, [](Value *V) { return match(V, m_AllowReassoc(m_Value())); });
+  bool AllowReassoc = all_of(VL, [&](Value *V) {
+    auto *I = dyn_cast<Instruction>(V);
+    if (!I || S.isCopyableElement(I))
+      return true;
+    return match(I, m_AllowReassoc(m_Value()));
+  });
   bool OnlyFirstOperand = AllowReassoc || S.getOpcode() == Instruction::FSub;
   unsigned NumCandidateOps =
       OnlyFirstOperand ? 1



More information about the llvm-commits mailing list