[llvm] [SLP] Let the caller tell canConvertToFMA which operand holds the fmul (PR #218754)

Dmitry Sidorov via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 9 06:53:45 PDT 2026


https://github.com/MrSidims updated https://github.com/llvm/llvm-project/pull/218754

>From 2eac97260692193b4e11071c05240e6feb47eea9 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sat, 22 Aug 2026 23:01:36 +0200
Subject: [PATCH 01/11] [SLP] Let the caller tell canConvertToFMA which operand
 holds the fmul

canConvertToFMA only ever looked at operand 0 of the fadd/fsub, so a
tree whose fmul sits on the right hand side was never recognized as
fusable.
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    |  39 +++--
 .../SLPVectorizer/X86/fma-operand-index.ll    |  26 ++--
 .../X86/horizontal-fadd-with-sub.ll           | 141 +++++++++++-------
 3 files changed, 128 insertions(+), 78 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 50d81058968d4..5ffa5884c323a 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -13653,7 +13653,18 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
                                        DominatorTree &DT, const DataLayout &DL,
                                        TargetTransformInfo &TTI,
                                        const TargetLibraryInfo &TLI,
-                                       const TTI::TargetCostKind CostKind);
+                                       const TTI::TargetCostKind CostKind,
+                                       unsigned FMulOpIdx = 0);
+
+/// \returns the operand of \p I that is a one-use fmul, 0 if there is none.
+static unsigned getFMulOperandIdx(const Instruction *I) {
+  for (unsigned Idx : seq<unsigned>(0, I->getNumOperands())) {
+    auto *Op = dyn_cast<Instruction>(I->getOperand(Idx));
+    if (Op && Op->getOpcode() == Instruction::FMul && Op->hasOneUse())
+      return Idx;
+  }
+  return 0;
+}
 
 uint64_t BoUpSLP::getNumScalarInsts(bool HasTreeLoop) {
   uint64_t Total = 0;
@@ -13734,7 +13745,7 @@ uint64_t BoUpSLP::getNumScalarInsts(bool HasTreeLoop) {
                      I->getOpcode() != Instruction::FSub))
             continue;
           if (canConvertToFMA(I, InstructionsState(I, I), *DT, *DL, *TTI, *TLI,
-                              CostKind)
+                              CostKind, getFMulOperandIdx(I))
                   .isValid()) {
             assert(Count > 0 && "Underflow in scalar inst count (fma)");
             --Count;
@@ -14457,7 +14468,8 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
                                        DominatorTree &DT, const DataLayout &DL,
                                        TargetTransformInfo &TTI,
                                        const TargetLibraryInfo &TLI,
-                                       const TTI::TargetCostKind CostKind) {
+                                       const TTI::TargetCostKind CostKind,
+                                       unsigned FMulOpIdx) {
   assert(all_of(VL,
                 [](Value *V) {
                   return V->getType()->getScalarType()->isFloatingPointTy();
@@ -14489,13 +14501,15 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
   InstructionsCompatibilityAnalysis Analysis(DT, DL, TTI, TLI);
   SmallVector<BoUpSLP::ValueList> Operands = Analysis.buildOperands(S, VL);
 
-  InstructionsState OpS = getSameOpcode(Operands.front(), TLI);
+  if (FMulOpIdx >= Operands.size())
+    return InstructionCost::getInvalid();
+  InstructionsState OpS = getSameOpcode(Operands[FMulOpIdx], TLI);
   if (!OpS.valid())
     return InstructionCost::getInvalid();
 
   if (OpS.isAltShuffle() || OpS.getOpcode() != Instruction::FMul)
     return InstructionCost::getInvalid();
-  if (!CheckForContractable(Operands.front(), OpS))
+  if (!CheckForContractable(Operands[FMulOpIdx], OpS))
     return InstructionCost::getInvalid();
   // Compare the costs.
   InstructionCost FMulPlusFAddCost = 0;
@@ -14536,7 +14550,7 @@ static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
         {I->getOperand(0), I->getOperand(1)});
   }
   unsigned NumOps = 0;
-  for (auto [V, Op] : zip(VL, Operands.front())) {
+  for (auto [V, Op] : zip(VL, Operands[FMulOpIdx])) {
     if (S.isCopyableElement(V))
       continue;
     auto *I = dyn_cast<Instruction>(Op);
@@ -17068,9 +17082,10 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
     return IntrinsicCost;
   };
   auto GetFMulAddCost = [&, &TTI = *TTI](const InstructionsState &S,
-                                         Instruction *VI) {
+                                         Instruction *VI,
+                                         unsigned FMulOpIdx = 0) {
     InstructionCost Cost =
-        canConvertToFMA(VI, S, *DT, *DL, TTI, *TLI, CostKind);
+        canConvertToFMA(VI, S, *DT, *DL, TTI, *TLI, CostKind, FMulOpIdx);
     return Cost;
   };
   switch (ShuffleOrOp) {
@@ -17708,7 +17723,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
       if (auto *I = dyn_cast<Instruction>(UniqueValues[Idx]);
           I && (ShuffleOrOp == Instruction::FAdd ||
                 ShuffleOrOp == Instruction::FSub)) {
-        InstructionCost IntrinsicCost = GetFMulAddCost(E->getOperations(), I);
+        InstructionCost IntrinsicCost =
+            GetFMulAddCost(E->getOperations(), I, getFMulOperandIdx(I));
         if (IntrinsicCost.isValid())
           ScalarCost = IntrinsicCost;
       }
@@ -33062,7 +33078,7 @@ bool SLPVectorizerPass::tryToVectorize(
       (I->getOpcode() == Instruction::FAdd ||
        I->getOpcode() == Instruction::FSub) &&
       canConvertToFMA(I, getSameOpcode(I, *TLI), *DT, *DL, *TTI, *TLI,
-                      R.getCostKind())
+                      R.getCostKind(), getFMulOperandIdx(I))
           .isValid()) {
     FMACandidates.insert(I);
     return false;
@@ -34338,7 +34354,8 @@ bool SLPVectorizerPass::vectorizeOnceUsedSeeds(BasicBlock *BB, BoUpSLP &R) {
       auto *U = cast<Instruction>(I.user_back());
       if (InstructionsState S = getSameOpcode(U, *TLI);
           S && S.isAddSubLikeOp() &&
-          canConvertToFMA(U, S, *DT, *DL, *TTI, *TLI, R.getCostKind())
+          canConvertToFMA(U, S, *DT, *DL, *TTI, *TLI, R.getCostKind(),
+                          U->getOperand(1) == &I ? 1 : 0)
               .isValid())
         continue;
     }
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
index abf8abf106eee..a5b28dd341e7a 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
@@ -10,16 +10,19 @@ define double @fmul_rhs_fadd(ptr %x, ptr %y, ptr %z) {
 ; CHECK-LABEL: define double @fmul_rhs_fadd(
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT:    [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
 ; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT:    [[X0:%.*]] = load double, ptr [[X]], align 8
+; CHECK-NEXT:    [[Y0:%.*]] = load double, ptr [[Y]], align 8
+; CHECK-NEXT:    [[MUL0:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
 ; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT:    [[X1:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT:    [[Y1:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT:    [[MUL1:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
 ; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
 ; CHECK-NEXT:    [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT:    [[MUL0:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
 ; CHECK-NEXT:    [[ADD0:%.*]] = fadd reassoc nsz contract double [[ZSUM]], [[MUL0]]
-; CHECK-NEXT:    [[MUL1:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
 ; CHECK-NEXT:    [[ADD1:%.*]] = fadd reassoc nsz contract double [[ADD0]], [[MUL1]]
 ; CHECK-NEXT:    ret double [[ADD1]]
 ;
@@ -46,16 +49,19 @@ define double @fmul_rhs_fsub(ptr %x, ptr %y, ptr %z) {
 ; CHECK-LABEL: define double @fmul_rhs_fsub(
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT:    [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
 ; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT:    [[X0:%.*]] = load double, ptr [[X]], align 8
+; CHECK-NEXT:    [[Y0:%.*]] = load double, ptr [[Y]], align 8
+; CHECK-NEXT:    [[MUL0:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
 ; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT:    [[X1:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT:    [[Y1:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT:    [[MUL1:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
 ; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
 ; CHECK-NEXT:    [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT:    [[MUL0:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
 ; CHECK-NEXT:    [[SUB0:%.*]] = fsub reassoc nsz contract double [[ZSUM]], [[MUL0]]
-; CHECK-NEXT:    [[MUL1:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
 ; CHECK-NEXT:    [[SUB1:%.*]] = fsub reassoc nsz contract double [[SUB0]], [[MUL1]]
 ; CHECK-NEXT:    ret double [[SUB1]]
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
index f96ab87e86049..853ab156bed97 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
@@ -9,16 +9,19 @@ define double @fsub_fmul_2(ptr %x, ptr %y, ptr %z) {
 ; CHECK-LABEL: define double @fsub_fmul_2(
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT:    [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
 ; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT:    [[X0:%.*]] = load double, ptr [[X]], align 8
+; CHECK-NEXT:    [[Y0:%.*]] = load double, ptr [[Y]], align 8
+; CHECK-NEXT:    [[TMP5:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
 ; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = load <2 x double>, ptr [[X]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
-; CHECK-NEXT:    [[TMP3:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[X1:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT:    [[Y1:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT:    [[TMP4:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
 ; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
 ; CHECK-NEXT:    [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <2 x double> [[TMP3]], i64 0
 ; CHECK-NEXT:    [[SUB:%.*]] = fsub reassoc nsz contract double [[TMP5]], [[ZSUM]]
-; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <2 x double> [[TMP3]], i64 1
 ; CHECK-NEXT:    [[TMP6:%.*]] = fadd reassoc nsz contract double [[SUB]], [[TMP4]]
 ; CHECK-NEXT:    ret double [[TMP6]]
 ;
@@ -46,25 +49,28 @@ define double @fsub_fmul_4(ptr %x, ptr %y, ptr %z) {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
 ; CHECK-NEXT:    [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
 ; CHECK-NEXT:    [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
+; CHECK-NEXT:    [[X16:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 16
+; CHECK-NEXT:    [[Y16:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 16
 ; CHECK-NEXT:    [[X24:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 24
 ; CHECK-NEXT:    [[Y24:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 24
 ; CHECK-NEXT:    [[X0:%.*]] = load double, ptr [[X]], align 8
 ; CHECK-NEXT:    [[Y0:%.*]] = load double, ptr [[Y]], align 8
 ; CHECK-NEXT:    [[MUL:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
-; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X8]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y8]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
-; CHECK-NEXT:    [[X3:%.*]] = load double, ptr [[X24]], align 8
-; CHECK-NEXT:    [[Y3:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT:    [[X3:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT:    [[Y3:%.*]] = load double, ptr [[Y8]], align 8
 ; CHECK-NEXT:    [[MUL16:%.*]] = fmul reassoc nsz contract double [[Y3]], [[X3]]
+; CHECK-NEXT:    [[X2:%.*]] = load double, ptr [[X16]], align 8
+; CHECK-NEXT:    [[Y2:%.*]] = load double, ptr [[Y16]], align 8
+; CHECK-NEXT:    [[TMP7:%.*]] = fmul reassoc nsz contract double [[Y2]], [[X2]]
+; CHECK-NEXT:    [[X4:%.*]] = load double, ptr [[X24]], align 8
+; CHECK-NEXT:    [[Y4:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT:    [[MUL17:%.*]] = fmul reassoc nsz contract double [[Y4]], [[X4]]
 ; CHECK-NEXT:    [[TMP3:%.*]] = load <4 x double>, ptr [[Z]], align 8
-; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
-; CHECK-NEXT:    [[T1:%.*]] = fadd reassoc nsz contract double [[MUL]], [[TMP4]]
-; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
+; CHECK-NEXT:    [[T1:%.*]] = fadd reassoc nsz contract double [[MUL]], [[MUL16]]
 ; CHECK-NEXT:    [[T3:%.*]] = fadd reassoc nsz contract double [[T1]], [[TMP7]]
 ; CHECK-NEXT:    [[TMP6:%.*]] = call reassoc nsz contract double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP3]])
 ; CHECK-NEXT:    [[ADD13:%.*]] = fsub reassoc nsz contract double [[T3]], [[TMP6]]
-; CHECK-NEXT:    [[TMP5:%.*]] = fadd reassoc nsz contract double [[ADD13]], [[MUL16]]
+; CHECK-NEXT:    [[TMP5:%.*]] = fadd reassoc nsz contract double [[ADD13]], [[MUL17]]
 ; CHECK-NEXT:    ret double [[TMP5]]
 ;
 entry:
@@ -390,30 +396,36 @@ define double @fsub_fmul_subtrahend_4(ptr %x, ptr %y, ptr %z) {
 ; CHECK-LABEL: define double @fsub_fmul_subtrahend_4(
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT:    [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
 ; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
 ; CHECK-NEXT:    [[X16:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 16
 ; CHECK-NEXT:    [[Y16:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 16
 ; CHECK-NEXT:    [[Z16:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 16
+; CHECK-NEXT:    [[X24:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 24
+; CHECK-NEXT:    [[Y24:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 24
 ; CHECK-NEXT:    [[Z24:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 24
+; CHECK-NEXT:    [[X0:%.*]] = load double, ptr [[X]], align 8
+; CHECK-NEXT:    [[Y0:%.*]] = load double, ptr [[Y]], align 8
+; CHECK-NEXT:    [[TMP6:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
 ; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT:    [[X1:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT:    [[Y1:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT:    [[TMP7:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
 ; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
+; CHECK-NEXT:    [[X2:%.*]] = load double, ptr [[X16]], align 8
+; CHECK-NEXT:    [[Y2:%.*]] = load double, ptr [[Y16]], align 8
+; CHECK-NEXT:    [[TMP8:%.*]] = fmul reassoc nsz contract double [[Y2]], [[X2]]
 ; CHECK-NEXT:    [[Z2:%.*]] = load double, ptr [[Z16]], align 8
-; CHECK-NEXT:    [[TMP3:%.*]] = load <2 x double>, ptr [[X16]], align 8
-; CHECK-NEXT:    [[TMP4:%.*]] = load <2 x double>, ptr [[Y16]], align 8
-; CHECK-NEXT:    [[TMP5:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP4]], [[TMP3]]
+; CHECK-NEXT:    [[X3:%.*]] = load double, ptr [[X24]], align 8
+; CHECK-NEXT:    [[Y3:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT:    [[TMP9:%.*]] = fmul reassoc nsz contract double [[Y3]], [[X3]]
 ; CHECK-NEXT:    [[Z3:%.*]] = load double, ptr [[Z24]], align 8
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
 ; CHECK-NEXT:    [[S0:%.*]] = fsub reassoc nsz contract double [[Z0]], [[TMP6]]
 ; CHECK-NEXT:    [[A1:%.*]] = fadd reassoc nsz contract double [[S0]], [[Z1]]
-; CHECK-NEXT:    [[TMP7:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
 ; CHECK-NEXT:    [[S1:%.*]] = fsub reassoc nsz contract double [[A1]], [[TMP7]]
 ; CHECK-NEXT:    [[A2:%.*]] = fadd reassoc nsz contract double [[S1]], [[Z2]]
-; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x double> [[TMP5]], i64 0
 ; CHECK-NEXT:    [[S2:%.*]] = fsub reassoc nsz contract double [[A2]], [[TMP8]]
-; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <2 x double> [[TMP5]], i64 1
 ; CHECK-NEXT:    [[S3:%.*]] = fsub reassoc nsz contract double [[S2]], [[TMP9]]
 ; CHECK-NEXT:    [[R:%.*]] = fadd reassoc nsz contract double [[S3]], [[Z3]]
 ; CHECK-NEXT:    ret double [[R]]
@@ -499,50 +511,59 @@ define double @three_sign_groups(ptr %x, ptr %y, ptr %u, ptr %v, ptr %z) {
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[U:%.*]], ptr [[V:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
 ; CHECK-NEXT:    [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT:    [[X16:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 16
 ; CHECK-NEXT:    [[X24:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 24
 ; CHECK-NEXT:    [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
+; CHECK-NEXT:    [[Y16:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 16
 ; CHECK-NEXT:    [[Y24:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 24
+; CHECK-NEXT:    [[U8:%.*]] = getelementptr inbounds nuw i8, ptr [[U]], i64 8
 ; CHECK-NEXT:    [[U16:%.*]] = getelementptr inbounds nuw i8, ptr [[U]], i64 16
+; CHECK-NEXT:    [[U24:%.*]] = getelementptr inbounds nuw i8, ptr [[U]], i64 24
+; CHECK-NEXT:    [[V8:%.*]] = getelementptr inbounds nuw i8, ptr [[V]], i64 8
 ; CHECK-NEXT:    [[V16:%.*]] = getelementptr inbounds nuw i8, ptr [[V]], i64 16
+; CHECK-NEXT:    [[V24:%.*]] = getelementptr inbounds nuw i8, ptr [[V]], i64 24
 ; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
 ; CHECK-NEXT:    [[Z16:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 16
 ; CHECK-NEXT:    [[Z24:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 24
 ; CHECK-NEXT:    [[X0:%.*]] = load double, ptr [[X]], align 8
-; CHECK-NEXT:    [[X3:%.*]] = load double, ptr [[X24]], align 8
+; CHECK-NEXT:    [[X3:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT:    [[X2:%.*]] = load double, ptr [[X16]], align 8
+; CHECK-NEXT:    [[X4:%.*]] = load double, ptr [[X24]], align 8
 ; CHECK-NEXT:    [[Y0:%.*]] = load double, ptr [[Y]], align 8
-; CHECK-NEXT:    [[Y3:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT:    [[Y3:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT:    [[Y2:%.*]] = load double, ptr [[Y16]], align 8
+; CHECK-NEXT:    [[Y4:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT:    [[U0:%.*]] = load double, ptr [[U]], align 8
+; CHECK-NEXT:    [[U1:%.*]] = load double, ptr [[U8]], align 8
+; CHECK-NEXT:    [[U2:%.*]] = load double, ptr [[U16]], align 8
+; CHECK-NEXT:    [[U3:%.*]] = load double, ptr [[U24]], align 8
+; CHECK-NEXT:    [[V0:%.*]] = load double, ptr [[V]], align 8
+; CHECK-NEXT:    [[V1:%.*]] = load double, ptr [[V8]], align 8
+; CHECK-NEXT:    [[V2:%.*]] = load double, ptr [[V16]], align 8
+; CHECK-NEXT:    [[V3:%.*]] = load double, ptr [[V24]], align 8
 ; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
 ; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
 ; CHECK-NEXT:    [[Z2:%.*]] = load double, ptr [[Z16]], align 8
 ; CHECK-NEXT:    [[Z3:%.*]] = load double, ptr [[Z24]], align 8
 ; CHECK-NEXT:    [[MXY0:%.*]] = fmul reassoc nsz contract double [[X0]], [[Y0]]
-; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X8]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y8]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP0]], [[TMP1]]
 ; CHECK-NEXT:    [[MXY3:%.*]] = fmul reassoc nsz contract double [[X3]], [[Y3]]
-; CHECK-NEXT:    [[TMP3:%.*]] = load <2 x double>, ptr [[U]], align 8
-; CHECK-NEXT:    [[TMP4:%.*]] = load <2 x double>, ptr [[V]], align 8
-; CHECK-NEXT:    [[TMP5:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP3]], [[TMP4]]
-; CHECK-NEXT:    [[TMP6:%.*]] = load <2 x double>, ptr [[U16]], align 8
-; CHECK-NEXT:    [[TMP7:%.*]] = load <2 x double>, ptr [[V16]], align 8
-; CHECK-NEXT:    [[TMP8:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP6]], [[TMP7]]
-; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
-; CHECK-NEXT:    [[T0:%.*]] = fadd reassoc nsz contract double [[MXY0]], [[TMP9]]
-; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
+; CHECK-NEXT:    [[TMP10:%.*]] = fmul reassoc nsz contract double [[X2]], [[Y2]]
+; CHECK-NEXT:    [[MXY4:%.*]] = fmul reassoc nsz contract double [[X4]], [[Y4]]
+; CHECK-NEXT:    [[TMP11:%.*]] = fmul reassoc nsz contract double [[U0]], [[V0]]
+; CHECK-NEXT:    [[TMP12:%.*]] = fmul reassoc nsz contract double [[U1]], [[V1]]
+; CHECK-NEXT:    [[TMP13:%.*]] = fmul reassoc nsz contract double [[U2]], [[V2]]
+; CHECK-NEXT:    [[TMP14:%.*]] = fmul reassoc nsz contract double [[U3]], [[V3]]
+; CHECK-NEXT:    [[T0:%.*]] = fadd reassoc nsz contract double [[MXY0]], [[MXY3]]
 ; CHECK-NEXT:    [[T1:%.*]] = fadd reassoc nsz contract double [[T0]], [[TMP10]]
-; CHECK-NEXT:    [[TMP11:%.*]] = extractelement <2 x double> [[TMP5]], i64 0
 ; CHECK-NEXT:    [[S0:%.*]] = fsub reassoc nsz contract double [[T1]], [[TMP11]]
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <2 x double> [[TMP5]], i64 1
 ; CHECK-NEXT:    [[S1:%.*]] = fsub reassoc nsz contract double [[S0]], [[TMP12]]
-; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <2 x double> [[TMP8]], i64 0
 ; CHECK-NEXT:    [[S2:%.*]] = fsub reassoc nsz contract double [[S1]], [[TMP13]]
-; CHECK-NEXT:    [[TMP14:%.*]] = extractelement <2 x double> [[TMP8]], i64 1
 ; CHECK-NEXT:    [[S3:%.*]] = fsub reassoc nsz contract double [[S2]], [[TMP14]]
 ; CHECK-NEXT:    [[W0:%.*]] = fsub reassoc nsz contract double [[S3]], [[Z0]]
 ; CHECK-NEXT:    [[W1:%.*]] = fsub reassoc nsz contract double [[W0]], [[Z1]]
 ; CHECK-NEXT:    [[W2:%.*]] = fsub reassoc nsz contract double [[W1]], [[Z2]]
 ; CHECK-NEXT:    [[W3:%.*]] = fsub reassoc nsz contract double [[W2]], [[Z3]]
-; CHECK-NEXT:    [[R:%.*]] = fadd reassoc nsz contract double [[W3]], [[MXY3]]
+; CHECK-NEXT:    [[R:%.*]] = fadd reassoc nsz contract double [[W3]], [[MXY4]]
 ; CHECK-NEXT:    ret double [[R]]
 ;
 entry:
@@ -611,30 +632,33 @@ define double @negated_repeated_pair(ptr %x, ptr %y, ptr %z) {
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
 ; CHECK-NEXT:    [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT:    [[X16:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 16
 ; CHECK-NEXT:    [[X24:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 24
 ; CHECK-NEXT:    [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
+; CHECK-NEXT:    [[Y16:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 16
 ; CHECK-NEXT:    [[Y24:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 24
 ; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
 ; CHECK-NEXT:    [[X0:%.*]] = load double, ptr [[X]], align 8
-; CHECK-NEXT:    [[X3:%.*]] = load double, ptr [[X24]], align 8
+; CHECK-NEXT:    [[X3:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT:    [[X2:%.*]] = load double, ptr [[X16]], align 8
+; CHECK-NEXT:    [[X4:%.*]] = load double, ptr [[X24]], align 8
 ; CHECK-NEXT:    [[Y0:%.*]] = load double, ptr [[Y]], align 8
-; CHECK-NEXT:    [[Y3:%.*]] = load double, ptr [[Y24]], align 8
+; CHECK-NEXT:    [[Y3:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT:    [[Y2:%.*]] = load double, ptr [[Y16]], align 8
+; CHECK-NEXT:    [[Y4:%.*]] = load double, ptr [[Y24]], align 8
 ; CHECK-NEXT:    [[ZA:%.*]] = load double, ptr [[Z]], align 8
 ; CHECK-NEXT:    [[ZB:%.*]] = load double, ptr [[Z8]], align 8
 ; CHECK-NEXT:    [[M0:%.*]] = fmul reassoc nsz contract double [[X0]], [[Y0]]
-; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X8]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y8]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP0]], [[TMP1]]
 ; CHECK-NEXT:    [[M3:%.*]] = fmul reassoc nsz contract double [[X3]], [[Y3]]
-; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
-; CHECK-NEXT:    [[T0:%.*]] = fadd reassoc nsz contract double [[M0]], [[TMP3]]
-; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
+; CHECK-NEXT:    [[TMP4:%.*]] = fmul reassoc nsz contract double [[X2]], [[Y2]]
+; CHECK-NEXT:    [[M4:%.*]] = fmul reassoc nsz contract double [[X4]], [[Y4]]
+; CHECK-NEXT:    [[T0:%.*]] = fadd reassoc nsz contract double [[M0]], [[M3]]
 ; CHECK-NEXT:    [[T1:%.*]] = fadd reassoc nsz contract double [[T0]], [[TMP4]]
 ; CHECK-NEXT:    [[S0:%.*]] = fsub reassoc nsz contract double [[T1]], [[ZA]]
 ; CHECK-NEXT:    [[S1:%.*]] = fsub reassoc nsz contract double [[S0]], [[ZB]]
 ; CHECK-NEXT:    [[S2:%.*]] = fsub reassoc nsz contract double [[S1]], [[ZA]]
 ; CHECK-NEXT:    [[S3:%.*]] = fsub reassoc nsz contract double [[S2]], [[ZB]]
-; CHECK-NEXT:    [[R:%.*]] = fadd reassoc nsz contract double [[S3]], [[M3]]
+; CHECK-NEXT:    [[R:%.*]] = fadd reassoc nsz contract double [[S3]], [[M4]]
 ; CHECK-NEXT:    ret double [[R]]
 ;
 entry:
@@ -707,17 +731,20 @@ define double @ordered_switch_restart(ptr %x, ptr %y, ptr %z, ptr %out) {
 ; CHECK-LABEL: define double @ordered_switch_restart(
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]], ptr [[OUT:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[X8:%.*]] = getelementptr inbounds nuw i8, ptr [[X]], i64 8
+; CHECK-NEXT:    [[Y8:%.*]] = getelementptr inbounds nuw i8, ptr [[Y]], i64 8
 ; CHECK-NEXT:    [[Z8:%.*]] = getelementptr inbounds nuw i8, ptr [[Z]], i64 8
+; CHECK-NEXT:    [[X0:%.*]] = load double, ptr [[X]], align 8
+; CHECK-NEXT:    [[Y0:%.*]] = load double, ptr [[Y]], align 8
+; CHECK-NEXT:    [[TMP3:%.*]] = fmul reassoc nsz contract double [[Y0]], [[X0]]
 ; CHECK-NEXT:    [[Z0:%.*]] = load double, ptr [[Z]], align 8
-; CHECK-NEXT:    [[TMP0:%.*]] = load <2 x double>, ptr [[X]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = load <2 x double>, ptr [[Y]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = fmul reassoc nsz contract <2 x double> [[TMP1]], [[TMP0]]
+; CHECK-NEXT:    [[X1:%.*]] = load double, ptr [[X8]], align 8
+; CHECK-NEXT:    [[Y1:%.*]] = load double, ptr [[Y8]], align 8
+; CHECK-NEXT:    [[TMP4:%.*]] = fmul reassoc nsz contract double [[Y1]], [[X1]]
 ; CHECK-NEXT:    [[Z1:%.*]] = load double, ptr [[Z8]], align 8
 ; CHECK-NEXT:    [[ZSUM:%.*]] = fadd reassoc nsz contract double [[Z0]], [[Z1]]
 ; CHECK-NEXT:    store double [[ZSUM]], ptr [[OUT]], align 8
-; CHECK-NEXT:    [[TMP3:%.*]] = extractelement <2 x double> [[TMP2]], i64 0
 ; CHECK-NEXT:    [[SUB:%.*]] = fsub reassoc nsz contract double [[TMP3]], [[ZSUM]]
-; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <2 x double> [[TMP2]], i64 1
 ; CHECK-NEXT:    [[ADD:%.*]] = fadd reassoc nsz contract double [[SUB]], [[TMP4]]
 ; CHECK-NEXT:    ret double [[ADD]]
 ;

>From 8bd4e20ff0a3a3a5651e29abc65517d7d4741973 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Tue, 25 Aug 2026 21:43:33 +0200
Subject: [PATCH 02/11] format

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 24 ++++++++-----------
 1 file changed, 10 insertions(+), 14 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 5ffa5884c323a..0a66731d45862 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -13648,13 +13648,11 @@ bool BoUpSLP::areAllUsersVectorized(
          });
 }
 
-static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
-                                       const InstructionsState &S,
-                                       DominatorTree &DT, const DataLayout &DL,
-                                       TargetTransformInfo &TTI,
-                                       const TargetLibraryInfo &TLI,
-                                       const TTI::TargetCostKind CostKind,
-                                       unsigned FMulOpIdx = 0);
+static InstructionCost
+canConvertToFMA(ArrayRef<Value *> VL, const InstructionsState &S,
+                DominatorTree &DT, const DataLayout &DL,
+                TargetTransformInfo &TTI, const TargetLibraryInfo &TLI,
+                const TTI::TargetCostKind CostKind, unsigned FMulOpIdx = 0);
 
 /// \returns the operand of \p I that is a one-use fmul, 0 if there is none.
 static unsigned getFMulOperandIdx(const Instruction *I) {
@@ -14463,13 +14461,11 @@ void BoUpSLP::reorderGatherNode(TreeEntry &TE) {
 
 /// Check if we can convert fadd/fsub sequence to FMAD.
 /// \returns Cost of the FMAD, if conversion is possible, invalid cost otherwise.
-static InstructionCost canConvertToFMA(ArrayRef<Value *> VL,
-                                       const InstructionsState &S,
-                                       DominatorTree &DT, const DataLayout &DL,
-                                       TargetTransformInfo &TTI,
-                                       const TargetLibraryInfo &TLI,
-                                       const TTI::TargetCostKind CostKind,
-                                       unsigned FMulOpIdx) {
+static InstructionCost
+canConvertToFMA(ArrayRef<Value *> VL, const InstructionsState &S,
+                DominatorTree &DT, const DataLayout &DL,
+                TargetTransformInfo &TTI, const TargetLibraryInfo &TLI,
+                const TTI::TargetCostKind CostKind, unsigned FMulOpIdx) {
   assert(all_of(VL,
                 [](Value *V) {
                   return V->getType()->getScalarType()->isFloatingPointTy();

>From 2b708b28959cfb7ff9a2cb43e38ba77846654631 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sat, 29 Aug 2026 23:01:17 +0200
Subject: [PATCH 03/11] partially address comments

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 25 ++++++-------------
 .../Vectorize/SLPVectorizer/SLPUtils.cpp      | 12 +++++++++
 .../Vectorize/SLPVectorizer/SLPUtils.h        |  4 +++
 3 files changed, 23 insertions(+), 18 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 3478f5064b416..1779611454340 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -13756,17 +13756,7 @@ static InstructionCost
 canConvertToFMA(ArrayRef<Value *> VL, const InstructionsState &S,
                 DominatorTree &DT, const DataLayout &DL,
                 TargetTransformInfo &TTI, const TargetLibraryInfo &TLI,
-                const TTI::TargetCostKind CostKind, unsigned FMulOpIdx = 0);
-
-/// \returns the operand of \p I that is a one-use fmul, 0 if there is none.
-static unsigned getFMulOperandIdx(const Instruction *I) {
-  for (unsigned Idx : seq<unsigned>(0, I->getNumOperands())) {
-    auto *Op = dyn_cast<Instruction>(I->getOperand(Idx));
-    if (Op && Op->getOpcode() == Instruction::FMul && Op->hasOneUse())
-      return Idx;
-  }
-  return 0;
-}
+                const TTI::TargetCostKind CostKind, unsigned FMulOpIdx);
 
 uint64_t BoUpSLP::getNumScalarInsts(bool HasTreeLoop) {
   uint64_t Total = 0;
@@ -15461,7 +15451,7 @@ void BoUpSLP::transformNodes() {
            !IsOneUseVectorFMulOperand(RHS)))
         break;
       if (!canConvertToFMA(E.Scalars, E.getOperations(), *DT, *DL, *TTI, *TLI,
-                           CostKind)
+                           CostKind, /*FMulOpIdx=*/0)
                .isValid())
         break;
       // This node is a fmuladd node.
@@ -17182,8 +17172,7 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
     return IntrinsicCost;
   };
   auto GetFMulAddCost = [&, &TTI = *TTI](const InstructionsState &S,
-                                         Instruction *VI,
-                                         unsigned FMulOpIdx = 0) {
+                                         Instruction *VI, unsigned FMulOpIdx) {
     InstructionCost Cost =
         canConvertToFMA(VI, S, *DT, *DL, TTI, *TLI, CostKind, FMulOpIdx);
     return Cost;
@@ -17649,8 +17638,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
     auto GetScalarCost = [&](unsigned Idx) {
       if (isa<PoisonValue>(UniqueValues[Idx]))
         return InstructionCost(TTI::TCC_Free);
-      return GetFMulAddCost(E->getOperations(),
-                            cast<Instruction>(UniqueValues[Idx]));
+      auto *VI = cast<Instruction>(UniqueValues[Idx]);
+      return GetFMulAddCost(E->getOperations(), VI, getFMulOperandIdx(VI));
     };
     auto GetVectorCost = [&, &TTI = *TTI](InstructionCost CommonCost) {
       FastMathFlags FMF;
@@ -32281,7 +32270,7 @@ class HorizontalReduction {
                 if (RdxKind == RecurKind::FAdd) {
                   InstructionCost FMACost =
                       canConvertToFMA(RdxOp, getSameOpcode(RdxOp, TLI), DT, DL,
-                                      *TTI, TLI, CostKind);
+                                      *TTI, TLI, CostKind, /*FMulOpIdx=*/0);
                   if (FMACost.isValid()) {
                     LLVM_DEBUG(dbgs() << "FMA cost: " << FMACost << "\n");
                     if (auto *I = dyn_cast<Instruction>(RdxVal)) {
@@ -32407,7 +32396,7 @@ class HorizontalReduction {
             }
             if (!Ops.empty()) {
               FMACost = canConvertToFMA(Ops, getSameOpcode(Ops, TLI), DT, DL,
-                                        *TTI, TLI, CostKind);
+                                        *TTI, TLI, CostKind, /*FMulOpIdx=*/0);
               if (FMACost.isValid()) {
                 // Calculate actual FMAD cost.
                 IntrinsicCostAttributes ICA(Intrinsic::fmuladd, RVecTy,
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index af6f8032c7b0b..d8ab0d5f1d3c9 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -771,6 +771,18 @@ Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable) {
   return Narrow;
 }
 
+unsigned getFMulOperandIdx(const Instruction *I) {
+  assert((I->getOpcode() == Instruction::Add ||
+          I->getOpcode() == Instruction::Sub ||
+          I->getOpcode() == Instruction::FAdd ||
+          I->getOpcode() == Instruction::FSub) &&
+         "Expected an add/sub-like instruction");
+  for (unsigned Idx : seq<unsigned>(I->getNumOperands()))
+    if (match(I->getOperand(Idx), m_OneUse(m_FMul(m_Value(), m_Value()))))
+      return Idx;
+  return 0;
+}
+
 namespace {
 
 /// Shifts and the mask accumulated from the narrow ops on the current path:
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index e9aeff605eb91..774cd02ada4f7 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -337,6 +337,10 @@ bool isOnceUsedSeed(const Instruction *I);
 /// produce nan/inf.
 Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable);
 
+/// \returns the operand index of \p I that holds a one-use fmul, 0 if there is
+/// none. \p I must be an add/sub-like instruction.
+unsigned getFMulOperandIdx(const Instruction *I);
+
 /// Narrow reduction leaf: the value, the shift applied after widening and
 /// the mask applied in the narrow type before widening, clearing the bits
 /// the absorbed narrow shls shift out and applying the absorbed narrow

>From 97c99be6ff7290a9671dc6452ec60a7e1483a2a1 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 30 Aug 2026 19:33:30 +0200
Subject: [PATCH 04/11] wip

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 24 ++++++++++---------
 .../Vectorize/SLPVectorizer/SLPUtils.cpp      |  6 ++---
 .../Vectorize/SLPVectorizer/SLPUtils.h        |  2 +-
 3 files changed, 16 insertions(+), 16 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 1779611454340..e20bffb02800c 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -14591,8 +14591,7 @@ canConvertToFMA(ArrayRef<Value *> VL, const InstructionsState &S,
   InstructionsCompatibilityAnalysis Analysis(DT, DL, TTI, TLI);
   SmallVector<BoUpSLP::ValueList> Operands = Analysis.buildOperands(S, VL);
 
-  if (FMulOpIdx >= Operands.size())
-    return InstructionCost::getInvalid();
+  assert(FMulOpIdx < Operands.size() && "Expected a binary add/sub-like op");
   InstructionsState OpS = getSameOpcode(Operands[FMulOpIdx], TLI);
   if (!OpS.valid())
     return InstructionCost::getInvalid();
@@ -15446,17 +15445,17 @@ void BoUpSLP::transformNodes() {
                         V->hasOneUse();
                });
       };
-      if (!IsOneUseVectorFMulOperand(LHS) &&
-          (E.getOpcode() == Instruction::FSub ||
-           !IsOneUseVectorFMulOperand(RHS)))
+      const bool FMulOnLHS = IsOneUseVectorFMulOperand(LHS);
+      if (!FMulOnLHS && (E.getOpcode() == Instruction::FSub ||
+                         !IsOneUseVectorFMulOperand(RHS)))
         break;
       if (!canConvertToFMA(E.Scalars, E.getOperations(), *DT, *DL, *TTI, *TLI,
-                           CostKind, /*FMulOpIdx=*/0)
+                           CostKind, FMulOnLHS ? 0 : 1)
                .isValid())
         break;
       // This node is a fmuladd node.
       E.CombinedOp = TreeEntry::FMulAdd;
-      TreeEntry *FMulEntry = getOperandEntry(&E, 0);
+      TreeEntry *FMulEntry = getOperandEntry(&E, FMulOnLHS ? 0 : 1);
       if (FMulEntry->UserTreeIndex &&
           FMulEntry->State == TreeEntry::Vectorize) {
         // The FMul node is part of the combined fmuladd node.
@@ -17639,7 +17638,9 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
       if (isa<PoisonValue>(UniqueValues[Idx]))
         return InstructionCost(TTI::TCC_Free);
       auto *VI = cast<Instruction>(UniqueValues[Idx]);
-      return GetFMulAddCost(E->getOperations(), VI, getFMulOperandIdx(VI));
+      return GetFMulAddCost(E->getOperations(), VI,
+                            E->isCopyableElement(VI) ? 0
+                                                     : getFMulOperandIdx(VI));
     };
     auto GetVectorCost = [&, &TTI = *TTI](InstructionCost CommonCost) {
       FastMathFlags FMF;
@@ -17810,8 +17811,9 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
       InstructionCost ScalarCost = TTI->getArithmeticInstrCost(
           ShuffleOrOp, OrigScalarTy, CostKind, Op1Info, Op2Info, Operands);
       if (auto *I = dyn_cast<Instruction>(UniqueValues[Idx]);
-          I && (ShuffleOrOp == Instruction::FAdd ||
-                ShuffleOrOp == Instruction::FSub)) {
+          I && !E->isCopyableElement(I) &&
+          (ShuffleOrOp == Instruction::FAdd ||
+           ShuffleOrOp == Instruction::FSub)) {
         InstructionCost IntrinsicCost =
             GetFMulAddCost(E->getOperations(), I, getFMulOperandIdx(I));
         if (IntrinsicCost.isValid())
@@ -34481,7 +34483,7 @@ bool SLPVectorizerPass::vectorizeOnceUsedSeeds(BasicBlock *BB, BoUpSLP &R) {
       if (InstructionsState S = getSameOpcode(U, *TLI);
           S && S.isAddSubLikeOp() &&
           canConvertToFMA(U, S, *DT, *DL, *TTI, *TLI, R.getCostKind(),
-                          U->getOperand(1) == &I ? 1 : 0)
+                          getFMulOperandIdx(U))
               .isValid())
         continue;
     }
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index d8ab0d5f1d3c9..960d49866ad2c 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -772,11 +772,9 @@ Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable) {
 }
 
 unsigned getFMulOperandIdx(const Instruction *I) {
-  assert((I->getOpcode() == Instruction::Add ||
-          I->getOpcode() == Instruction::Sub ||
-          I->getOpcode() == Instruction::FAdd ||
+  assert((I->getOpcode() == Instruction::FAdd ||
           I->getOpcode() == Instruction::FSub) &&
-         "Expected an add/sub-like instruction");
+         "Expected an fadd/fsub-like instruction");
   for (unsigned Idx : seq<unsigned>(I->getNumOperands()))
     if (match(I->getOperand(Idx), m_OneUse(m_FMul(m_Value(), m_Value()))))
       return Idx;
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
index 774cd02ada4f7..9edc8eda3cd8d 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.h
@@ -338,7 +338,7 @@ bool isOnceUsedSeed(const Instruction *I);
 Instruction *lookThroughCastRoundTrip(Value *V, bool MustBeElidable);
 
 /// \returns the operand index of \p I that holds a one-use fmul, 0 if there is
-/// none. \p I must be an add/sub-like instruction.
+/// none. \p I must be an fadd/fsub-like instruction.
 unsigned getFMulOperandIdx(const Instruction *I);
 
 /// Narrow reduction leaf: the value, the shift applied after widening and

>From 0f178028f239fca03d10a40ff25cfacadd47c95f Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 30 Aug 2026 22:58:20 +0200
Subject: [PATCH 05/11] update tests

---
 .../AMDGPU/elementwise-fma-operand1.ll        | 126 +++++++++++++++---
 .../X86/select-logical-or-and-i1-vector.ll    | 102 +++++++-------
 2 files changed, 162 insertions(+), 66 deletions(-)

diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
index c4c97622ac726..c928eb4df3227 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
@@ -18,12 +18,42 @@ define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr
 ; CHECK-LABEL: define void @axpy4_contract(
 ; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = fmul contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP4:%.*]] = fadd contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT:    store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT:    [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT:    [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT:    [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT:    [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT:    [[R0:%.*]] = fadd contract float [[C0]], [[M0]]
+; CHECK-NEXT:    store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT:    [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT:    [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT:    [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT:    [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT:    [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT:    [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT:    [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT:    [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; CHECK-NEXT:    [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT:    store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT:    [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT:    [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT:    [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT:    [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT:    [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT:    [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT:    [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; CHECK-NEXT:    [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; CHECK-NEXT:    [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT:    store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT:    [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT:    [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT:    [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT:    [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT:    [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT:    [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT:    [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; CHECK-NEXT:    [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; CHECK-NEXT:    [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT:    store float [[R3]], ptr [[DP3]], align 4
 ; CHECK-NEXT:    ret void
 ;
 ; THR12-LABEL: define void @axpy4_contract(
@@ -81,12 +111,42 @@ define void @axpy4_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr n
 ; CHECK-LABEL: define void @axpy4_reassoc(
 ; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = fmul reassoc contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP4:%.*]] = fadd reassoc contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT:    store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT:    [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT:    [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT:    [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT:    [[M0:%.*]] = fmul reassoc contract float [[A0]], [[B0]]
+; CHECK-NEXT:    [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; CHECK-NEXT:    store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT:    [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT:    [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT:    [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT:    [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT:    [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT:    [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT:    [[M1:%.*]] = fmul reassoc contract float [[A1]], [[B1]]
+; CHECK-NEXT:    [[R1:%.*]] = fadd reassoc contract float [[C1]], [[M1]]
+; CHECK-NEXT:    [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT:    store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT:    [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT:    [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT:    [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT:    [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT:    [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT:    [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT:    [[M2:%.*]] = fmul reassoc contract float [[A2]], [[B2]]
+; CHECK-NEXT:    [[R2:%.*]] = fadd reassoc contract float [[C2]], [[M2]]
+; CHECK-NEXT:    [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT:    store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT:    [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT:    [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT:    [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT:    [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT:    [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT:    [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT:    [[M3:%.*]] = fmul reassoc contract float [[A3]], [[B3]]
+; CHECK-NEXT:    [[R3:%.*]] = fadd reassoc contract float [[C3]], [[M3]]
+; CHECK-NEXT:    [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT:    store float [[R3]], ptr [[DP3]], align 4
 ; CHECK-NEXT:    ret void
 ;
 ; THR12-LABEL: define void @axpy4_reassoc(
@@ -144,12 +204,42 @@ define void @axpy4_mixed_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b,
 ; CHECK-LABEL: define void @axpy4_mixed_reassoc(
 ; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = fmul contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP4:%.*]] = fadd contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT:    store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT:    [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT:    [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT:    [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT:    [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT:    [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; CHECK-NEXT:    store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT:    [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT:    [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT:    [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT:    [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT:    [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT:    [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT:    [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT:    [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; CHECK-NEXT:    [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT:    store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT:    [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT:    [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT:    [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT:    [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT:    [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT:    [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT:    [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; CHECK-NEXT:    [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; CHECK-NEXT:    [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT:    store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT:    [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT:    [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT:    [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT:    [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT:    [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT:    [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT:    [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; CHECK-NEXT:    [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; CHECK-NEXT:    [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT:    store float [[R3]], ptr [[DP3]], align 4
 ; CHECK-NEXT:    ret void
 ;
 ; THR12-LABEL: define void @axpy4_mixed_reassoc(
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll b/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
index 979701609b499..bb10829b8cca2 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
@@ -12,30 +12,33 @@ define void @select_logical_or_i1(ptr %dst,
 ; CHECK-LABEL: define void @select_logical_or_i1(
 ; CHECK-SAME: ptr [[DST:%.*]], float [[D0:%.*]], float [[D1:%.*]], float [[D2:%.*]], float [[D3:%.*]], float [[THRESHOLD:%.*]], float [[HPHB_VAL:%.*]], i1 [[SCALAR_COND:%.*]], float [[Y0:%.*]], float [[Y1:%.*]], float [[Y2:%.*]], float [[Y3:%.*]], float [[E0:%.*]], float [[E1:%.*]], float [[E2:%.*]], float [[E3:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x float> poison, float [[D0]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x float> [[TMP0]], float [[D1]], i64 1
-; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x float> [[TMP1]], float [[D2]], i64 2
-; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x float> [[TMP2]], float [[D3]], i64 3
-; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x float> poison, float [[THRESHOLD]], i64 0
-; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <4 x float> [[TMP4]], <4 x float> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP6:%.*]] = fcmp fast uge <4 x float> [[TMP3]], [[TMP5]]
-; CHECK-NEXT:    [[TMP7:%.*]] = insertelement <4 x i1> poison, i1 [[SCALAR_COND]], i64 0
-; CHECK-NEXT:    [[TMP8:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP9:%.*]] = select <4 x i1> [[TMP6]], <4 x i1> splat (i1 true), <4 x i1> [[TMP8]]
-; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <4 x float> poison, float [[HPHB_VAL]], i64 0
-; CHECK-NEXT:    [[TMP11:%.*]] = shufflevector <4 x float> [[TMP10]], <4 x float> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x float> zeroinitializer, <4 x float> [[TMP11]]
-; CHECK-NEXT:    [[TMP13:%.*]] = insertelement <4 x float> poison, float [[Y0]], i64 0
-; CHECK-NEXT:    [[TMP14:%.*]] = insertelement <4 x float> [[TMP13]], float [[Y1]], i64 1
-; CHECK-NEXT:    [[TMP15:%.*]] = insertelement <4 x float> [[TMP14]], float [[Y2]], i64 2
-; CHECK-NEXT:    [[TMP16:%.*]] = insertelement <4 x float> [[TMP15]], float [[Y3]], i64 3
-; CHECK-NEXT:    [[TMP17:%.*]] = fmul fast <4 x float> [[TMP12]], [[TMP16]]
-; CHECK-NEXT:    [[TMP18:%.*]] = insertelement <4 x float> poison, float [[E0]], i64 0
-; CHECK-NEXT:    [[TMP19:%.*]] = insertelement <4 x float> [[TMP18]], float [[E1]], i64 1
-; CHECK-NEXT:    [[TMP20:%.*]] = insertelement <4 x float> [[TMP19]], float [[E2]], i64 2
-; CHECK-NEXT:    [[TMP21:%.*]] = insertelement <4 x float> [[TMP20]], float [[E3]], i64 3
-; CHECK-NEXT:    [[TMP22:%.*]] = fadd fast <4 x float> [[TMP21]], [[TMP17]]
-; CHECK-NEXT:    store <4 x float> [[TMP22]], ptr [[DST]], align 4
+; CHECK-NEXT:    [[CMP0:%.*]] = fcmp fast uge float [[D0]], [[THRESHOLD]]
+; CHECK-NEXT:    [[CMP1:%.*]] = fcmp fast uge float [[D1]], [[THRESHOLD]]
+; CHECK-NEXT:    [[CMP2:%.*]] = fcmp fast uge float [[D2]], [[THRESHOLD]]
+; CHECK-NEXT:    [[CMP3:%.*]] = fcmp fast uge float [[D3]], [[THRESHOLD]]
+; CHECK-NEXT:    [[OR0:%.*]] = select i1 [[CMP0]], i1 true, i1 [[SCALAR_COND]]
+; CHECK-NEXT:    [[OR1:%.*]] = select i1 [[CMP1]], i1 true, i1 [[SCALAR_COND]]
+; CHECK-NEXT:    [[OR2:%.*]] = select i1 [[CMP2]], i1 true, i1 [[SCALAR_COND]]
+; CHECK-NEXT:    [[OR3:%.*]] = select i1 [[CMP3]], i1 true, i1 [[SCALAR_COND]]
+; CHECK-NEXT:    [[SEL0:%.*]] = select i1 [[OR0]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT:    [[SEL1:%.*]] = select i1 [[OR1]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT:    [[SEL2:%.*]] = select i1 [[OR2]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT:    [[SEL3:%.*]] = select i1 [[OR3]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT:    [[MUL0:%.*]] = fmul fast float [[SEL0]], [[Y0]]
+; CHECK-NEXT:    [[MUL1:%.*]] = fmul fast float [[SEL1]], [[Y1]]
+; CHECK-NEXT:    [[MUL2:%.*]] = fmul fast float [[SEL2]], [[Y2]]
+; CHECK-NEXT:    [[MUL3:%.*]] = fmul fast float [[SEL3]], [[Y3]]
+; CHECK-NEXT:    [[RES0:%.*]] = fadd fast float [[E0]], [[MUL0]]
+; CHECK-NEXT:    [[RES1:%.*]] = fadd fast float [[E1]], [[MUL1]]
+; CHECK-NEXT:    [[RES2:%.*]] = fadd fast float [[E2]], [[MUL2]]
+; CHECK-NEXT:    [[RES3:%.*]] = fadd fast float [[E3]], [[MUL3]]
+; CHECK-NEXT:    store float [[RES0]], ptr [[DST]], align 4
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 1
+; CHECK-NEXT:    store float [[RES1]], ptr [[P1]], align 4
+; CHECK-NEXT:    [[P2:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 2
+; CHECK-NEXT:    store float [[RES2]], ptr [[P2]], align 4
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 3
+; CHECK-NEXT:    store float [[RES3]], ptr [[P3]], align 4
 ; CHECK-NEXT:    ret void
 ;
   float %d0, float %d1, float %d2, float %d3,
@@ -84,30 +87,33 @@ define void @select_logical_and_i1(ptr %dst,
 ; CHECK-LABEL: define void @select_logical_and_i1(
 ; CHECK-SAME: ptr [[DST:%.*]], float [[D0:%.*]], float [[D1:%.*]], float [[D2:%.*]], float [[D3:%.*]], float [[THRESHOLD:%.*]], float [[HPHB_VAL:%.*]], i1 [[SCALAR_COND:%.*]], float [[Y0:%.*]], float [[Y1:%.*]], float [[Y2:%.*]], float [[Y3:%.*]], float [[E0:%.*]], float [[E1:%.*]], float [[E2:%.*]], float [[E3:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <4 x float> poison, float [[D0]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <4 x float> [[TMP0]], float [[D1]], i64 1
-; CHECK-NEXT:    [[TMP2:%.*]] = insertelement <4 x float> [[TMP1]], float [[D2]], i64 2
-; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x float> [[TMP2]], float [[D3]], i64 3
-; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x float> poison, float [[THRESHOLD]], i64 0
-; CHECK-NEXT:    [[TMP5:%.*]] = shufflevector <4 x float> [[TMP4]], <4 x float> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP6:%.*]] = fcmp fast uge <4 x float> [[TMP3]], [[TMP5]]
-; CHECK-NEXT:    [[TMP7:%.*]] = insertelement <4 x i1> poison, i1 [[SCALAR_COND]], i64 0
-; CHECK-NEXT:    [[TMP8:%.*]] = shufflevector <4 x i1> [[TMP7]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP9:%.*]] = select <4 x i1> [[TMP6]], <4 x i1> [[TMP8]], <4 x i1> zeroinitializer
-; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <4 x float> poison, float [[HPHB_VAL]], i64 0
-; CHECK-NEXT:    [[TMP11:%.*]] = shufflevector <4 x float> [[TMP10]], <4 x float> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x float> zeroinitializer, <4 x float> [[TMP11]]
-; CHECK-NEXT:    [[TMP13:%.*]] = insertelement <4 x float> poison, float [[Y0]], i64 0
-; CHECK-NEXT:    [[TMP14:%.*]] = insertelement <4 x float> [[TMP13]], float [[Y1]], i64 1
-; CHECK-NEXT:    [[TMP15:%.*]] = insertelement <4 x float> [[TMP14]], float [[Y2]], i64 2
-; CHECK-NEXT:    [[TMP16:%.*]] = insertelement <4 x float> [[TMP15]], float [[Y3]], i64 3
-; CHECK-NEXT:    [[TMP17:%.*]] = fmul fast <4 x float> [[TMP12]], [[TMP16]]
-; CHECK-NEXT:    [[TMP18:%.*]] = insertelement <4 x float> poison, float [[E0]], i64 0
-; CHECK-NEXT:    [[TMP19:%.*]] = insertelement <4 x float> [[TMP18]], float [[E1]], i64 1
-; CHECK-NEXT:    [[TMP20:%.*]] = insertelement <4 x float> [[TMP19]], float [[E2]], i64 2
-; CHECK-NEXT:    [[TMP21:%.*]] = insertelement <4 x float> [[TMP20]], float [[E3]], i64 3
-; CHECK-NEXT:    [[TMP22:%.*]] = fadd fast <4 x float> [[TMP21]], [[TMP17]]
-; CHECK-NEXT:    store <4 x float> [[TMP22]], ptr [[DST]], align 4
+; CHECK-NEXT:    [[CMP0:%.*]] = fcmp fast uge float [[D0]], [[THRESHOLD]]
+; CHECK-NEXT:    [[CMP1:%.*]] = fcmp fast uge float [[D1]], [[THRESHOLD]]
+; CHECK-NEXT:    [[CMP2:%.*]] = fcmp fast uge float [[D2]], [[THRESHOLD]]
+; CHECK-NEXT:    [[CMP3:%.*]] = fcmp fast uge float [[D3]], [[THRESHOLD]]
+; CHECK-NEXT:    [[AND0:%.*]] = select i1 [[CMP0]], i1 [[SCALAR_COND]], i1 false
+; CHECK-NEXT:    [[AND1:%.*]] = select i1 [[CMP1]], i1 [[SCALAR_COND]], i1 false
+; CHECK-NEXT:    [[AND2:%.*]] = select i1 [[CMP2]], i1 [[SCALAR_COND]], i1 false
+; CHECK-NEXT:    [[AND3:%.*]] = select i1 [[CMP3]], i1 [[SCALAR_COND]], i1 false
+; CHECK-NEXT:    [[SEL0:%.*]] = select i1 [[AND0]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT:    [[SEL1:%.*]] = select i1 [[AND1]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT:    [[SEL2:%.*]] = select i1 [[AND2]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT:    [[SEL3:%.*]] = select i1 [[AND3]], float 0.000000e+00, float [[HPHB_VAL]]
+; CHECK-NEXT:    [[MUL0:%.*]] = fmul fast float [[SEL0]], [[Y0]]
+; CHECK-NEXT:    [[MUL1:%.*]] = fmul fast float [[SEL1]], [[Y1]]
+; CHECK-NEXT:    [[MUL2:%.*]] = fmul fast float [[SEL2]], [[Y2]]
+; CHECK-NEXT:    [[MUL3:%.*]] = fmul fast float [[SEL3]], [[Y3]]
+; CHECK-NEXT:    [[RES0:%.*]] = fadd fast float [[E0]], [[MUL0]]
+; CHECK-NEXT:    [[RES1:%.*]] = fadd fast float [[E1]], [[MUL1]]
+; CHECK-NEXT:    [[RES2:%.*]] = fadd fast float [[E2]], [[MUL2]]
+; CHECK-NEXT:    [[RES3:%.*]] = fadd fast float [[E3]], [[MUL3]]
+; CHECK-NEXT:    store float [[RES0]], ptr [[DST]], align 4
+; CHECK-NEXT:    [[P1:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 1
+; CHECK-NEXT:    store float [[RES1]], ptr [[P1]], align 4
+; CHECK-NEXT:    [[P2:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 2
+; CHECK-NEXT:    store float [[RES2]], ptr [[P2]], align 4
+; CHECK-NEXT:    [[P3:%.*]] = getelementptr inbounds float, ptr [[DST]], i64 3
+; CHECK-NEXT:    store float [[RES3]], ptr [[P3]], align 4
 ; CHECK-NEXT:    ret void
 ;
   float %d0, float %d1, float %d2, float %d3,

>From 6f8fbaa91013d053a96196f4527a4a6b56faa864 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Sun, 30 Aug 2026 23:57:27 +0200
Subject: [PATCH 06/11] update test

---
 .../AArch64/reassociate-fma-pairs.ll          | 34 +++++++++----------
 1 file changed, 17 insertions(+), 17 deletions(-)

diff --git a/llvm/test/Transforms/PhaseOrdering/AArch64/reassociate-fma-pairs.ll b/llvm/test/Transforms/PhaseOrdering/AArch64/reassociate-fma-pairs.ll
index 33df34fd80713..0a48a6a0f1015 100644
--- a/llvm/test/Transforms/PhaseOrdering/AArch64/reassociate-fma-pairs.ll
+++ b/llvm/test/Transforms/PhaseOrdering/AArch64/reassociate-fma-pairs.ll
@@ -14,49 +14,49 @@ define double @md_vdw_energy(ptr nocapture readonly %coeffs, double %energy, dou
 ; CHECK-NEXT:    [[FACTOR_OP_FMUL2:%.*]] = fmul fast double [[TABLE_DELTA]], 5.000000e-01
 ; CHECK-NEXT:    [[FACTOR_OP_FMUL3:%.*]] = fmul fast double [[FACTOR_OP_FMUL]], [[TABLE_DELTA]]
 ; CHECK-NEXT:    [[FACTOR_OP_FMUL4:%.*]] = fmul fast double [[TMP0]], 2.500000e-01
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x double> poison, double [[SCALE]], i64 0
-; CHECK-NEXT:    [[TMP11:%.*]] = shufflevector <2 x double> [[TMP1]], <2 x double> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[NEXT:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[ACC1:%.*]] = phi double [ [[ENERGY]], %[[ENTRY]] ], [ [[RESULT1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[BASE:%.*]] = getelementptr inbounds [8 x i8], ptr [[COEFFS]], i64 [[I]]
+; CHECK-NEXT:    [[A:%.*]] = load double, ptr [[BASE]], align 8
+; CHECK-NEXT:    [[B_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 8
+; CHECK-NEXT:    [[B:%.*]] = load double, ptr [[B_PTR]], align 8
+; CHECK-NEXT:    [[TMP5:%.*]] = fmul fast double [[A]], [[SCALE]]
+; CHECK-NEXT:    [[TMP6:%.*]] = fmul fast double [[B]], [[SCALE]]
 ; CHECK-NEXT:    [[C0_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 16
 ; CHECK-NEXT:    [[C0:%.*]] = load double, ptr [[C0_PTR]], align 8
 ; CHECK-NEXT:    [[C1_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 24
 ; CHECK-NEXT:    [[C1:%.*]] = load double, ptr [[C1_PTR]], align 8
+; CHECK-NEXT:    [[P0:%.*]] = fmul fast double [[C0]], [[TMP5]]
+; CHECK-NEXT:    [[Q0:%.*]] = fmul fast double [[C1]], [[TMP6]]
+; CHECK-NEXT:    [[D2:%.*]] = fsub fast double [[P0]], [[Q0]]
 ; CHECK-NEXT:    [[C2_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 32
 ; CHECK-NEXT:    [[C2:%.*]] = load double, ptr [[C2_PTR]], align 8
 ; CHECK-NEXT:    [[C3_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 40
 ; CHECK-NEXT:    [[C3:%.*]] = load double, ptr [[C3_PTR]], align 8
+; CHECK-NEXT:    [[P1:%.*]] = fmul fast double [[C2]], [[TMP5]]
+; CHECK-NEXT:    [[Q1:%.*]] = fmul fast double [[C3]], [[TMP6]]
+; CHECK-NEXT:    [[D1:%.*]] = fsub fast double [[P1]], [[Q1]]
 ; CHECK-NEXT:    [[C4_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 48
 ; CHECK-NEXT:    [[C4:%.*]] = load double, ptr [[C4_PTR]], align 8
 ; CHECK-NEXT:    [[C5_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 56
 ; CHECK-NEXT:    [[C5:%.*]] = load double, ptr [[C5_PTR]], align 8
-; CHECK-NEXT:    [[C6_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 64
-; CHECK-NEXT:    [[TMP3:%.*]] = load <2 x double>, ptr [[BASE]], align 8
-; CHECK-NEXT:    [[TMP4:%.*]] = fmul fast <2 x double> [[TMP3]], [[TMP11]]
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <2 x double> [[TMP4]], i64 0
-; CHECK-NEXT:    [[P2:%.*]] = fmul fast double [[C0]], [[TMP5]]
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <2 x double> [[TMP4]], i64 1
-; CHECK-NEXT:    [[Q2:%.*]] = fmul fast double [[C1]], [[TMP6]]
-; CHECK-NEXT:    [[D2:%.*]] = fsub fast double [[P2]], [[Q2]]
-; CHECK-NEXT:    [[P1:%.*]] = fmul fast double [[C2]], [[TMP5]]
-; CHECK-NEXT:    [[Q1:%.*]] = fmul fast double [[C3]], [[TMP6]]
-; CHECK-NEXT:    [[D1:%.*]] = fsub fast double [[P1]], [[Q1]]
 ; CHECK-NEXT:    [[ACC:%.*]] = fmul fast double [[C4]], [[TMP5]]
 ; CHECK-NEXT:    [[TMP2:%.*]] = fmul fast double [[C5]], [[TMP6]]
 ; CHECK-NEXT:    [[D3_NEG:%.*]] = fsub fast double [[ACC]], [[TMP2]]
-; CHECK-NEXT:    [[TMP7:%.*]] = load <2 x double>, ptr [[C6_PTR]], align 8
+; CHECK-NEXT:    [[C6_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 64
+; CHECK-NEXT:    [[C6:%.*]] = load double, ptr [[C6_PTR]], align 8
+; CHECK-NEXT:    [[C7_PTR:%.*]] = getelementptr inbounds nuw i8, ptr [[BASE]], i64 72
+; CHECK-NEXT:    [[C7:%.*]] = load double, ptr [[C7_PTR]], align 8
 ; CHECK-NEXT:    [[T1_REASS_REASS:%.*]] = fmul fast double [[FACTOR_OP_FMUL3]], [[D2]]
 ; CHECK-NEXT:    [[T2_REASS_REASS:%.*]] = fmul fast double [[D1]], [[FACTOR_OP_FMUL4]]
 ; CHECK-NEXT:    [[S1:%.*]] = fadd fast double [[T2_REASS_REASS]], [[T1_REASS_REASS]]
 ; CHECK-NEXT:    [[S2_NEG:%.*]] = fmul fast double [[D3_NEG]], [[FACTOR_OP_FMUL2]]
 ; CHECK-NEXT:    [[RESULT:%.*]] = fadd fast double [[S1]], [[S2_NEG]]
-; CHECK-NEXT:    [[TMP8:%.*]] = fmul fast <2 x double> [[TMP7]], [[TMP4]]
-; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <2 x double> [[TMP8]], i64 0
+; CHECK-NEXT:    [[TMP9:%.*]] = fmul fast double [[C6]], [[TMP5]]
+; CHECK-NEXT:    [[TMP10:%.*]] = fmul fast double [[C7]], [[TMP6]]
 ; CHECK-NEXT:    [[REASS_ADD:%.*]] = fadd fast double [[RESULT]], [[TMP9]]
-; CHECK-NEXT:    [[TMP10:%.*]] = extractelement <2 x double> [[TMP8]], i64 1
 ; CHECK-NEXT:    [[S2_NEG1:%.*]] = fadd fast double [[ACC1]], [[TMP10]]
 ; CHECK-NEXT:    [[RESULT1]] = fsub fast double [[S2_NEG1]], [[REASS_ADD]]
 ; CHECK-NEXT:    [[NEXT]] = add nuw i64 [[I]], 10

>From 71dc2c44588efea3080c4722184c989f09c13bfd Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Mon, 31 Aug 2026 17:42:11 +0200
Subject: [PATCH 07/11] update stale test comments

---
 .../SLPVectorizer/X86/fma-operand-index.ll           | 12 +++++++-----
 .../SLPVectorizer/X86/horizontal-fadd-with-sub.ll    |  5 +++--
 .../X86/select-logical-or-and-i1-vector.ll           |  4 ++++
 3 files changed, 14 insertions(+), 7 deletions(-)

diff --git a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
index a5b28dd341e7a..b2fa19d65e2e2 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/fma-operand-index.ll
@@ -1,11 +1,13 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
 ; RUN: opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu -mcpu=znver4 < %s | FileCheck %s
 
-; A one-use fmul is left scalar so the backend can fuse it, but only operand 0
-; of the fadd/fsub is checked. The same fmul is kept or gathered depending on
-; which side it sits on. To be addressed in a follow-up.
+; A one-use fmul is left scalar so the backend can fuse it, whichever operand
+; of the fadd/fsub it sits on.
+;
+; FIXME: The scalar form gives up the wide x and y loads. #218755 makes the
+; veto depend on whether the target always profits from a scalar fma.
 
-; Both fmuls are operand 1, so they gather and cannot fuse.
+; Both fmuls are operand 1 and stay scalar to fuse.
 define double @fmul_rhs_fadd(ptr %x, ptr %y, ptr %z) {
 ; CHECK-LABEL: define double @fmul_rhs_fadd(
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0:[0-9]+]] {
@@ -83,7 +85,7 @@ entry:
   ret double %sub1
 }
 
-; Operand 0 is the one checked, so these stay scalar and fuse.
+; Fmuls on operand 0 likewise stay scalar and fuse.
 define double @fmul_lhs_fadd(ptr %x, ptr %y, ptr %z) {
 ; CHECK-LABEL: define double @fmul_lhs_fadd(
 ; CHECK-SAME: ptr [[X:%.*]], ptr [[Y:%.*]], ptr [[Z:%.*]]) #[[ATTR0]] {
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
index 853ab156bed97..119df94ee12c2 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/horizontal-fadd-with-sub.ll
@@ -2,8 +2,9 @@
 ; RUN: opt -S --passes=slp-vectorizer -mtriple=x86_64-unknown-linux-gnu -mcpu=znver4 < %s | FileCheck %s
 
 ; An fadd reduction over an fsub/fneg chain is flattened with per-operand
-; signs, so subtracted leaves are vectorized and subtracted in the final
-; combine, forming per-lane fma.
+; signs, so subtracted leaves can be vectorized and subtracted in the final
+; combine, forming per-lane fma. Where the leaves are one-use fmuls the
+; operand aware fma pricing may keep the chain scalar instead.
 
 define double @fsub_fmul_2(ptr %x, ptr %y, ptr %z) {
 ; CHECK-LABEL: define double @fsub_fmul_2(
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll b/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
index bb10829b8cca2..c94193c7fc592 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/select-logical-or-and-i1-vector.ll
@@ -7,6 +7,10 @@
 ; Reduced from a real-world molecular docking kernel where independent
 ; scalar fcmps are combined with a loop-invariant scalar condition via
 ; logical select, feeding into float selects stored to consecutive memory.
+;
+; FIXME: The operand aware fma pricing keeps the one-use fmuls scalar and the
+; rest of the tree with them, so nothing is vectorized right now. Recover the
+; vectorization without breaking the scalar fma pricing.
 
 define void @select_logical_or_i1(ptr %dst,
 ; CHECK-LABEL: define void @select_logical_or_i1(

>From 0a69f7777b41a4f1f0bbc39148f0c300af894aa1 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Mon, 31 Aug 2026 21:34:53 +0200
Subject: [PATCH 08/11] pass real fmul operand index at reduction sites

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 21 +++--
 .../X86/redux-feed-buildvector.ll             | 92 ++++++++++++++++---
 2 files changed, 96 insertions(+), 17 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index b5fdb50fb7879..37c7284eea118 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -32280,9 +32280,9 @@ class HorizontalReduction {
               auto *RdxOp = cast<Instruction>(U);
               if (hasRequiredNumberOfUses(IsCmpSelMinMax, RdxOp)) {
                 if (RdxKind == RecurKind::FAdd) {
-                  InstructionCost FMACost =
-                      canConvertToFMA(RdxOp, getSameOpcode(RdxOp, TLI), DT, DL,
-                                      *TTI, TLI, CostKind, /*FMulOpIdx=*/0);
+                  InstructionCost FMACost = canConvertToFMA(
+                      RdxOp, getSameOpcode(RdxOp, TLI), DT, DL, *TTI, TLI,
+                      CostKind, RdxOp->getOperand(1) == RdxVal ? 1 : 0);
                   if (FMACost.isValid()) {
                     LLVM_DEBUG(dbgs() << "FMA cost: " << FMACost << "\n");
                     if (auto *I = dyn_cast<Instruction>(RdxVal)) {
@@ -32397,18 +32397,27 @@ class HorizontalReduction {
             SmallVector<Value *> Ops;
             FastMathFlags FMF;
             FMF.set();
-            for (Value *RdxVal : ReducedVals) {
+            unsigned FMulOpIdx = 0;
+            for (auto [Idx, RdxVal] : enumerate(ReducedVals)) {
               if (!RdxVal->hasOneUse()) {
                 Ops.clear();
                 break;
               }
+              User *U = RdxVal->user_back();
+              unsigned OpIdx = U->getOperand(1) == RdxVal ? 1 : 0;
+              if (Idx == 0)
+                FMulOpIdx = OpIdx;
+              else if (FMulOpIdx != OpIdx) {
+                Ops.clear();
+                break;
+              }
               if (auto *FPCI = dyn_cast<FPMathOperator>(RdxVal))
                 FMF &= FPCI->getFastMathFlags();
-              Ops.push_back(RdxVal->user_back());
+              Ops.push_back(U);
             }
             if (!Ops.empty()) {
               FMACost = canConvertToFMA(Ops, getSameOpcode(Ops, TLI), DT, DL,
-                                        *TTI, TLI, CostKind, /*FMulOpIdx=*/0);
+                                        *TTI, TLI, CostKind, FMulOpIdx);
               if (FMACost.isValid()) {
                 // Calculate actual FMAD cost.
                 IntrinsicCostAttributes ICA(Intrinsic::fmuladd, RVecTy,
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/redux-feed-buildvector.ll b/llvm/test/Transforms/SLPVectorizer/X86/redux-feed-buildvector.ll
index 67054d202d5a2..73670db825f2e 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/redux-feed-buildvector.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/redux-feed-buildvector.ll
@@ -4,23 +4,93 @@
 ; The test represents the case with multiple vectorization possibilities
 ; but the most effective way to vectorize it is to match both 8-way reductions
 ; feeding the insertelement vector build sequence.
+;
+; FIXME: The operand aware fma pricing values the scalar form as 16 fmas and
+; declines both reductions, giving up the wide loads. Recover the reduction
+; vectorization here.
 
 declare void @llvm.masked.scatter.v2f64.v2p0(<2 x double>, <2 x ptr>, i32 immarg, <2 x i1>)
 
 define void @test(ptr nocapture readonly %arg, ptr nocapture readonly %arg1, ptr nocapture %arg2) {
 ; CHECK-LABEL: @test(
 ; CHECK-NEXT:  entry:
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <8 x ptr> poison, ptr [[ARG:%.*]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = shufflevector <8 x ptr> [[TMP0]], <8 x ptr> poison, <8 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP5:%.*]] = getelementptr inbounds double, <8 x ptr> [[TMP1]], <8 x i64> <i64 1, i64 3, i64 5, i64 7, i64 9, i64 11, i64 13, i64 15>
-; CHECK-NEXT:    [[GEP2_0:%.*]] = getelementptr inbounds double, ptr [[ARG1:%.*]], i64 16
-; CHECK-NEXT:    [[TMP2:%.*]] = call <8 x double> @llvm.masked.gather.v8f64.v8p0(<8 x ptr> align 8 [[TMP5]], <8 x i1> splat (i1 true), <8 x double> poison)
-; CHECK-NEXT:    [[TMP3:%.*]] = load <8 x double>, ptr [[GEP2_0]], align 8
-; CHECK-NEXT:    [[TMP4:%.*]] = fmul fast <8 x double> [[TMP3]], [[TMP2]]
-; CHECK-NEXT:    [[TMP6:%.*]] = load <8 x double>, ptr [[ARG1]], align 8
-; CHECK-NEXT:    [[TMP8:%.*]] = fmul fast <8 x double> [[TMP6]], [[TMP2]]
-; CHECK-NEXT:    [[TMP7:%.*]] = call fast double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> [[TMP8]])
-; CHECK-NEXT:    [[TMP11:%.*]] = call fast double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> [[TMP4]])
+; CHECK-NEXT:    [[GEP1_0:%.*]] = getelementptr inbounds double, ptr [[ARG:%.*]], i64 1
+; CHECK-NEXT:    [[LD1_0:%.*]] = load double, ptr [[GEP1_0]], align 8
+; CHECK-NEXT:    [[LD0_0:%.*]] = load double, ptr [[ARG1:%.*]], align 8
+; CHECK-NEXT:    [[MUL1_0:%.*]] = fmul fast double [[LD0_0]], [[LD1_0]]
+; CHECK-NEXT:    [[GEP2_0:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 16
+; CHECK-NEXT:    [[LD2_0:%.*]] = load double, ptr [[GEP2_0]], align 8
+; CHECK-NEXT:    [[MUL2_0:%.*]] = fmul fast double [[LD2_0]], [[LD1_0]]
+; CHECK-NEXT:    [[GEP1_1:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 3
+; CHECK-NEXT:    [[LD1_1:%.*]] = load double, ptr [[GEP1_1]], align 8
+; CHECK-NEXT:    [[GEP0_1:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 1
+; CHECK-NEXT:    [[LD0_1:%.*]] = load double, ptr [[GEP0_1]], align 8
+; CHECK-NEXT:    [[MUL1_1:%.*]] = fmul fast double [[LD0_1]], [[LD1_1]]
+; CHECK-NEXT:    [[RDX1_0:%.*]] = fadd fast double [[MUL1_0]], [[MUL1_1]]
+; CHECK-NEXT:    [[GEP2_1:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 17
+; CHECK-NEXT:    [[LD2_1:%.*]] = load double, ptr [[GEP2_1]], align 8
+; CHECK-NEXT:    [[MUL2_1:%.*]] = fmul fast double [[LD2_1]], [[LD1_1]]
+; CHECK-NEXT:    [[RDX2_0:%.*]] = fadd fast double [[MUL2_0]], [[MUL2_1]]
+; CHECK-NEXT:    [[GEP1_2:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 5
+; CHECK-NEXT:    [[LD1_2:%.*]] = load double, ptr [[GEP1_2]], align 8
+; CHECK-NEXT:    [[GEP0_2:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 2
+; CHECK-NEXT:    [[LD0_2:%.*]] = load double, ptr [[GEP0_2]], align 8
+; CHECK-NEXT:    [[MUL1_2:%.*]] = fmul fast double [[LD0_2]], [[LD1_2]]
+; CHECK-NEXT:    [[RDX1_1:%.*]] = fadd fast double [[RDX1_0]], [[MUL1_2]]
+; CHECK-NEXT:    [[GEP2_2:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 18
+; CHECK-NEXT:    [[LD2_2:%.*]] = load double, ptr [[GEP2_2]], align 8
+; CHECK-NEXT:    [[MUL2_2:%.*]] = fmul fast double [[LD2_2]], [[LD1_2]]
+; CHECK-NEXT:    [[RDX2_1:%.*]] = fadd fast double [[RDX2_0]], [[MUL2_2]]
+; CHECK-NEXT:    [[GEP1_3:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 7
+; CHECK-NEXT:    [[LD1_3:%.*]] = load double, ptr [[GEP1_3]], align 8
+; CHECK-NEXT:    [[GEP0_3:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 3
+; CHECK-NEXT:    [[LD0_3:%.*]] = load double, ptr [[GEP0_3]], align 8
+; CHECK-NEXT:    [[MUL1_3:%.*]] = fmul fast double [[LD0_3]], [[LD1_3]]
+; CHECK-NEXT:    [[RDX1_2:%.*]] = fadd fast double [[RDX1_1]], [[MUL1_3]]
+; CHECK-NEXT:    [[GEP2_3:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 19
+; CHECK-NEXT:    [[LD2_3:%.*]] = load double, ptr [[GEP2_3]], align 8
+; CHECK-NEXT:    [[MUL2_3:%.*]] = fmul fast double [[LD2_3]], [[LD1_3]]
+; CHECK-NEXT:    [[RDX2_2:%.*]] = fadd fast double [[RDX2_1]], [[MUL2_3]]
+; CHECK-NEXT:    [[GEP1_4:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 9
+; CHECK-NEXT:    [[LD1_4:%.*]] = load double, ptr [[GEP1_4]], align 8
+; CHECK-NEXT:    [[GEP0_4:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 4
+; CHECK-NEXT:    [[LD0_4:%.*]] = load double, ptr [[GEP0_4]], align 8
+; CHECK-NEXT:    [[MUL1_4:%.*]] = fmul fast double [[LD0_4]], [[LD1_4]]
+; CHECK-NEXT:    [[RDX1_3:%.*]] = fadd fast double [[RDX1_2]], [[MUL1_4]]
+; CHECK-NEXT:    [[GEP2_4:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 20
+; CHECK-NEXT:    [[LD2_4:%.*]] = load double, ptr [[GEP2_4]], align 8
+; CHECK-NEXT:    [[MUL2_4:%.*]] = fmul fast double [[LD2_4]], [[LD1_4]]
+; CHECK-NEXT:    [[RDX2_3:%.*]] = fadd fast double [[RDX2_2]], [[MUL2_4]]
+; CHECK-NEXT:    [[GEP1_5:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 11
+; CHECK-NEXT:    [[LD1_5:%.*]] = load double, ptr [[GEP1_5]], align 8
+; CHECK-NEXT:    [[GEP0_5:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 5
+; CHECK-NEXT:    [[LD0_5:%.*]] = load double, ptr [[GEP0_5]], align 8
+; CHECK-NEXT:    [[MUL1_5:%.*]] = fmul fast double [[LD0_5]], [[LD1_5]]
+; CHECK-NEXT:    [[RDX1_4:%.*]] = fadd fast double [[RDX1_3]], [[MUL1_5]]
+; CHECK-NEXT:    [[GEP2_5:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 21
+; CHECK-NEXT:    [[LD2_5:%.*]] = load double, ptr [[GEP2_5]], align 8
+; CHECK-NEXT:    [[MUL2_5:%.*]] = fmul fast double [[LD2_5]], [[LD1_5]]
+; CHECK-NEXT:    [[RDX2_4:%.*]] = fadd fast double [[RDX2_3]], [[MUL2_5]]
+; CHECK-NEXT:    [[GEP1_6:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 13
+; CHECK-NEXT:    [[LD1_6:%.*]] = load double, ptr [[GEP1_6]], align 8
+; CHECK-NEXT:    [[GEP0_6:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 6
+; CHECK-NEXT:    [[LD0_6:%.*]] = load double, ptr [[GEP0_6]], align 8
+; CHECK-NEXT:    [[MUL1_6:%.*]] = fmul fast double [[LD0_6]], [[LD1_6]]
+; CHECK-NEXT:    [[RDX1_5:%.*]] = fadd fast double [[RDX1_4]], [[MUL1_6]]
+; CHECK-NEXT:    [[GEP2_6:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 22
+; CHECK-NEXT:    [[LD2_6:%.*]] = load double, ptr [[GEP2_6]], align 8
+; CHECK-NEXT:    [[MUL2_6:%.*]] = fmul fast double [[LD2_6]], [[LD1_6]]
+; CHECK-NEXT:    [[RDX2_5:%.*]] = fadd fast double [[RDX2_4]], [[MUL2_6]]
+; CHECK-NEXT:    [[GEP1_7:%.*]] = getelementptr inbounds double, ptr [[ARG]], i64 15
+; CHECK-NEXT:    [[LD1_7:%.*]] = load double, ptr [[GEP1_7]], align 8
+; CHECK-NEXT:    [[GEP0_7:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 7
+; CHECK-NEXT:    [[LD0_7:%.*]] = load double, ptr [[GEP0_7]], align 8
+; CHECK-NEXT:    [[MUL1_7:%.*]] = fmul fast double [[LD0_7]], [[LD1_7]]
+; CHECK-NEXT:    [[TMP7:%.*]] = fadd fast double [[RDX1_5]], [[MUL1_7]]
+; CHECK-NEXT:    [[GEP2_7:%.*]] = getelementptr inbounds double, ptr [[ARG1]], i64 23
+; CHECK-NEXT:    [[LD2_7:%.*]] = load double, ptr [[GEP2_7]], align 8
+; CHECK-NEXT:    [[MUL2_7:%.*]] = fmul fast double [[LD2_7]], [[LD1_7]]
+; CHECK-NEXT:    [[TMP11:%.*]] = fadd fast double [[RDX2_5]], [[MUL2_7]]
 ; CHECK-NEXT:    [[I142:%.*]] = insertelement <2 x double> poison, double [[TMP7]], i64 0
 ; CHECK-NEXT:    [[I143:%.*]] = insertelement <2 x double> [[I142]], double [[TMP11]], i64 1
 ; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds double, ptr [[ARG2:%.*]], <2 x i64> <i64 0, i64 16>

>From e86a361377a2a462b5b52913e6933a4baf1e1928 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Wed, 2 Sep 2026 00:15:39 +0200
Subject: [PATCH 09/11] apply review

---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 23 ++++++++++++-------
 .../Vectorize/SLPVectorizer/SLPUtils.cpp      |  4 ++--
 2 files changed, 17 insertions(+), 10 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index d72e7dd3eef1f..3a0ba2c724919 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -32457,10 +32457,12 @@ class HorizontalReduction {
             for (User *U : RdxVal->users()) {
               auto *RdxOp = cast<Instruction>(U);
               if (hasRequiredNumberOfUses(IsCmpSelMinMax, RdxOp)) {
-                if (RdxKind == RecurKind::FAdd) {
-                  InstructionCost FMACost = canConvertToFMA(
-                      RdxOp, getSameOpcode(RdxOp, TLI), DT, DL, *TTI, TLI,
-                      CostKind, RdxOp->getOperand(1) == RdxVal ? 1 : 0);
+                auto *BO = dyn_cast<BinaryOperator>(RdxOp);
+                if (BO && RdxKind == RecurKind::FAdd) {
+                  unsigned OpIdx = BO->getOperand(1) == RdxVal ? 1 : 0;
+                  InstructionCost FMACost =
+                      canConvertToFMA(BO, getSameOpcode(BO, TLI), DT, DL, *TTI,
+                                      TLI, CostKind, OpIdx);
                   if (FMACost.isValid()) {
                     LLVM_DEBUG(dbgs() << "FMA cost: " << FMACost << "\n");
                     if (auto *I = dyn_cast<Instruction>(RdxVal)) {
@@ -32582,13 +32584,18 @@ class HorizontalReduction {
                 break;
               }
               User *U = RdxVal->user_back();
-              unsigned OpIdx = U->getOperand(1) == RdxVal ? 1 : 0;
-              if (Idx == 0)
-                FMulOpIdx = OpIdx;
-              else if (FMulOpIdx != OpIdx) {
+              auto *BO = dyn_cast<BinaryOperator>(U);
+              if (!BO) {
                 Ops.clear();
                 break;
               }
+              unsigned OpIdx = BO->getOperand(1) == RdxVal ? 1 : 0;
+              if (Idx != 0 && FMulOpIdx != OpIdx) {
+                Ops.clear();
+                break;
+              }
+              if (Idx == 0)
+                FMulOpIdx = OpIdx;
               if (auto *FPCI = dyn_cast<FPMathOperator>(RdxVal))
                 FMF &= FPCI->getFastMathFlags();
               Ops.push_back(U);
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
index 960d49866ad2c..9c3da31db33c9 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPUtils.cpp
@@ -775,8 +775,8 @@ unsigned getFMulOperandIdx(const Instruction *I) {
   assert((I->getOpcode() == Instruction::FAdd ||
           I->getOpcode() == Instruction::FSub) &&
          "Expected an fadd/fsub-like instruction");
-  for (unsigned Idx : seq<unsigned>(I->getNumOperands()))
-    if (match(I->getOperand(Idx), m_OneUse(m_FMul(m_Value(), m_Value()))))
+  for (auto [Idx, Op] : enumerate(I->operand_values()))
+    if (match(Op, m_OneUse(m_FMul(m_Value(), m_Value()))))
       return Idx;
   return 0;
 }

>From 82693e96d2a00f03a4ed0c85c49dbf3f2b4da2f3 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Wed, 9 Sep 2026 13:21:07 +0200
Subject: [PATCH 10/11] [SLP] Guard the reduction FMA cost on a valid
 add/sub-like state

The two reduction call sites built an InstructionsState and handed it straight
to canConvertToFMA. A reduction operand can be a non add/sub-like user, and the
state can be invalid, so check both before pricing the fusion.

Assisted-by: Claude Code Opus 5
---
 llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp | 16 ++++++++++------
 1 file changed, 10 insertions(+), 6 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 3a0ba2c724919..ecd6f418234bf 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -32458,11 +32458,13 @@ class HorizontalReduction {
               auto *RdxOp = cast<Instruction>(U);
               if (hasRequiredNumberOfUses(IsCmpSelMinMax, RdxOp)) {
                 auto *BO = dyn_cast<BinaryOperator>(RdxOp);
-                if (BO && RdxKind == RecurKind::FAdd) {
+                InstructionsState RdxOpS = BO && RdxKind == RecurKind::FAdd
+                                               ? getSameOpcode(BO, TLI)
+                                               : InstructionsState::invalid();
+                if (RdxOpS && RdxOpS.isAddSubLikeOp()) {
                   unsigned OpIdx = BO->getOperand(1) == RdxVal ? 1 : 0;
-                  InstructionCost FMACost =
-                      canConvertToFMA(BO, getSameOpcode(BO, TLI), DT, DL, *TTI,
-                                      TLI, CostKind, OpIdx);
+                  InstructionCost FMACost = canConvertToFMA(
+                      BO, RdxOpS, DT, DL, *TTI, TLI, CostKind, OpIdx);
                   if (FMACost.isValid()) {
                     LLVM_DEBUG(dbgs() << "FMA cost: " << FMACost << "\n");
                     if (auto *I = dyn_cast<Instruction>(RdxVal)) {
@@ -32601,8 +32603,10 @@ class HorizontalReduction {
               Ops.push_back(U);
             }
             if (!Ops.empty()) {
-              FMACost = canConvertToFMA(Ops, getSameOpcode(Ops, TLI), DT, DL,
-                                        *TTI, TLI, CostKind, FMulOpIdx);
+              InstructionsState S = getSameOpcode(Ops, TLI);
+              if (S && S.isAddSubLikeOp())
+                FMACost = canConvertToFMA(Ops, S, DT, DL, *TTI, TLI, CostKind,
+                                          FMulOpIdx);
               if (FMACost.isValid()) {
                 // Calculate actual FMAD cost.
                 IntrinsicCostAttributes ICA(Intrinsic::fmuladd, RVecTy,

>From f20af57e8910f159666558d56c50a76776f7c182 Mon Sep 17 00:00:00 2001
From: Dmitry Sidorov <Dmitry.Sidorov at amd.com>
Date: Wed, 9 Sep 2026 14:14:54 +0200
Subject: [PATCH 11/11] update test

---
 .../AArch64/hor-fp-reduction-instcount.ll     | 49 ++++++++++++-------
 1 file changed, 31 insertions(+), 18 deletions(-)

diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/hor-fp-reduction-instcount.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/hor-fp-reduction-instcount.ll
index 304db13bf6bb0..4a33c7cb55631 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/hor-fp-reduction-instcount.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/hor-fp-reduction-instcount.ll
@@ -2,46 +2,59 @@
 ; RUN: opt < %s -S -passes=slp-vectorizer -mtriple=aarch64-unknown-linux -mcpu=neoverse-v2 | FileCheck %s --check-prefix=NARROW
 ; RUN: opt < %s -S -passes=slp-vectorizer -mtriple=aarch64-unknown-linux -mcpu=neoverse-v1 | FileCheck %s --check-prefix=WIDE
 
-; A horizontal double reduction with gathered operands is not profitable on
-; targets whose vector register holds only 2 doubles: the operand gathering and
-; the reduction epilogue produce more instructions than the scalar reduction.
-; On wider targets (4+ doubles per register) it stays profitable.
+; A horizontal double reduction with gathered operands is not profitable. The
+; gathering and the reduction epilogue cost more than the scalar chain, whose
+; fadd links all fuse into fmadd.
 
 define double @f64_red_gather(ptr %a, ptr %w) {
 ; NARROW-LABEL: define double @f64_red_gather(
 ; NARROW-SAME: ptr [[A:%.*]], ptr [[W:%.*]]) #[[ATTR0:[0-9]+]] {
 ; NARROW-NEXT:    [[A0:%.*]] = load double, ptr [[A]], align 8
+; NARROW-NEXT:    [[W0:%.*]] = load double, ptr [[W]], align 8
+; NARROW-NEXT:    [[P0:%.*]] = fmul reassoc contract double [[A0]], [[W0]]
 ; NARROW-NEXT:    [[A1P:%.*]] = getelementptr double, ptr [[A]], i64 3
 ; NARROW-NEXT:    [[A1:%.*]] = load double, ptr [[A1P]], align 8
+; NARROW-NEXT:    [[W1P:%.*]] = getelementptr double, ptr [[W]], i64 1
+; NARROW-NEXT:    [[W1:%.*]] = load double, ptr [[W1P]], align 8
+; NARROW-NEXT:    [[P1:%.*]] = fmul reassoc contract double [[A1]], [[W1]]
 ; NARROW-NEXT:    [[A2P:%.*]] = getelementptr double, ptr [[A]], i64 6
 ; NARROW-NEXT:    [[A2:%.*]] = load double, ptr [[A2P]], align 8
+; NARROW-NEXT:    [[W2P:%.*]] = getelementptr double, ptr [[W]], i64 2
+; NARROW-NEXT:    [[W2:%.*]] = load double, ptr [[W2P]], align 8
+; NARROW-NEXT:    [[P2:%.*]] = fmul reassoc contract double [[A2]], [[W2]]
 ; NARROW-NEXT:    [[A3P:%.*]] = getelementptr double, ptr [[A]], i64 9
 ; NARROW-NEXT:    [[A3:%.*]] = load double, ptr [[A3P]], align 8
-; NARROW-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[W]], align 8
-; NARROW-NEXT:    [[TMP2:%.*]] = insertelement <4 x double> poison, double [[A0]], i64 0
-; NARROW-NEXT:    [[TMP3:%.*]] = insertelement <4 x double> [[TMP2]], double [[A1]], i64 1
-; NARROW-NEXT:    [[TMP4:%.*]] = insertelement <4 x double> [[TMP3]], double [[A2]], i64 2
-; NARROW-NEXT:    [[TMP5:%.*]] = insertelement <4 x double> [[TMP4]], double [[A3]], i64 3
-; NARROW-NEXT:    [[TMP6:%.*]] = fmul reassoc contract <4 x double> [[TMP5]], [[TMP1]]
-; NARROW-NEXT:    [[TMP7:%.*]] = call reassoc contract double @llvm.vector.reduce.fadd.v4f64(double -0.000000e+00, <4 x double> [[TMP6]])
+; NARROW-NEXT:    [[W3P:%.*]] = getelementptr double, ptr [[W]], i64 3
+; NARROW-NEXT:    [[W3:%.*]] = load double, ptr [[W3P]], align 8
+; NARROW-NEXT:    [[P3:%.*]] = fmul reassoc contract double [[A3]], [[W3]]
+; NARROW-NEXT:    [[R0:%.*]] = fadd reassoc contract double [[P0]], [[P1]]
+; NARROW-NEXT:    [[R1:%.*]] = fadd reassoc contract double [[R0]], [[P2]]
+; NARROW-NEXT:    [[TMP7:%.*]] = fadd reassoc contract double [[R1]], [[P3]]
 ; NARROW-NEXT:    ret double [[TMP7]]
 ;
 ; WIDE-LABEL: define double @f64_red_gather(
 ; WIDE-SAME: ptr [[A:%.*]], ptr [[W:%.*]]) #[[ATTR0:[0-9]+]] {
 ; WIDE-NEXT:    [[A0:%.*]] = load double, ptr [[A]], align 8
+; WIDE-NEXT:    [[W0:%.*]] = load double, ptr [[W]], align 8
+; WIDE-NEXT:    [[P0:%.*]] = fmul reassoc contract double [[A0]], [[W0]]
 ; WIDE-NEXT:    [[A1P:%.*]] = getelementptr double, ptr [[A]], i64 3
 ; WIDE-NEXT:    [[A1:%.*]] = load double, ptr [[A1P]], align 8
+; WIDE-NEXT:    [[W1P:%.*]] = getelementptr double, ptr [[W]], i64 1
+; WIDE-NEXT:    [[W1:%.*]] = load double, ptr [[W1P]], align 8
+; WIDE-NEXT:    [[P1:%.*]] = fmul reassoc contract double [[A1]], [[W1]]
 ; WIDE-NEXT:    [[A2P:%.*]] = getelementptr double, ptr [[A]], i64 6
 ; WIDE-NEXT:    [[A2:%.*]] = load double, ptr [[A2P]], align 8
+; WIDE-NEXT:    [[W2P:%.*]] = getelementptr double, ptr [[W]], i64 2
+; WIDE-NEXT:    [[W2:%.*]] = load double, ptr [[W2P]], align 8
+; WIDE-NEXT:    [[P2:%.*]] = fmul reassoc contract double [[A2]], [[W2]]
 ; WIDE-NEXT:    [[A3P:%.*]] = getelementptr double, ptr [[A]], i64 9
 ; WIDE-NEXT:    [[A3:%.*]] = load double, ptr [[A3P]], align 8
-; WIDE-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[W]], align 8
-; WIDE-NEXT:    [[TMP2:%.*]] = insertelement <4 x double> poison, double [[A0]], i64 0
-; WIDE-NEXT:    [[TMP3:%.*]] = insertelement <4 x double> [[TMP2]], double [[A1]], i64 1
-; WIDE-NEXT:    [[TMP4:%.*]] = insertelement <4 x double> [[TMP3]], double [[A2]], i64 2
-; WIDE-NEXT:    [[TMP5:%.*]] = insertelement <4 x double> [[TMP4]], double [[A3]], i64 3
-; WIDE-NEXT:    [[TMP6:%.*]] = fmul reassoc contract <4 x double> [[TMP5]], [[TMP1]]
-; WIDE-NEXT:    [[TMP7:%.*]] = call reassoc contract double @llvm.vector.reduce.fadd.v4f64(double -0.000000e+00, <4 x double> [[TMP6]])
+; WIDE-NEXT:    [[W3P:%.*]] = getelementptr double, ptr [[W]], i64 3
+; WIDE-NEXT:    [[W3:%.*]] = load double, ptr [[W3P]], align 8
+; WIDE-NEXT:    [[P3:%.*]] = fmul reassoc contract double [[A3]], [[W3]]
+; WIDE-NEXT:    [[R0:%.*]] = fadd reassoc contract double [[P0]], [[P1]]
+; WIDE-NEXT:    [[R1:%.*]] = fadd reassoc contract double [[R0]], [[P2]]
+; WIDE-NEXT:    [[TMP7:%.*]] = fadd reassoc contract double [[R1]], [[P3]]
 ; WIDE-NEXT:    ret double [[TMP7]]
 ;
   %a0 = load double, ptr %a



More information about the llvm-commits mailing list