[llvm] [SLP]Pass operand info and context to TTI cost queries (PR #224931)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 04:39:52 PDT 2026


https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/224931

>From e11e18def061b20e53f35a58e199c7c8a0aef243 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sun, 20 Sep 2026 06:54:43 -0700
Subject: [PATCH 1/2] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20in?=
 =?UTF-8?q?itial=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 198 +++++++++++-------
 .../SLPVectorizer/SLPMemoryUtils.cpp          |   3 +-
 .../AArch64/externally-used-copyables.ll      |  13 +-
 .../AMDGPU/elementwise-fma-operand1.ll        | 129 ++++++++++--
 ...on-commutative-second-arg-only-copyable.ll |  10 +-
 .../RISCV/reduced-copyable-element.ll         |   6 +-
 .../strided-loads-with-external-indices.ll    |  12 +-
 .../SystemZ/non-power-2-subvector-extract.ll  |  36 +++-
 .../SLPVectorizer/X86/phi-multi-same-nodes.ll |   5 +-
 9 files changed, 290 insertions(+), 122 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 205311d10e76c..26075b1ef2a87 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -6391,11 +6391,11 @@ BoUpSLP::LoadsState BoUpSLP::canVectorizeLoads(
         }
         switch (LS) {
         case LoadsState::Vectorize:
-          VecLdCost +=
-              TTI.getMemoryOpCost(Instruction::Load, SubVecTy, LI0->getAlign(),
-                                  LI0->getPointerAddressSpace(), CostKind,
-                                  TTI::OperandValueInfo()) +
-              VectorGEPCost;
+          VecLdCost += TTI.getMemoryOpCost(
+                           Instruction::Load, SubVecTy, LI0->getAlign(),
+                           LI0->getPointerAddressSpace(), CostKind,
+                           TTI::getOperandInfo(LI0->getPointerOperand())) +
+                       VectorGEPCost;
           break;
         case LoadsState::StridedVectorize:
           VecLdCost += TTI.getMemIntrinsicInstrCost(
@@ -9086,7 +9086,10 @@ static InstructionCost getVectorOpCost(Instruction *I, unsigned VF,
         getVectorCallCosts(CI, VecTy, &TTI, &TLI, ArgTys, CostKind);
     return std::min(IntrCost, LibCost);
   }
-  return TTI.getArithmeticInstrCost(I->getOpcode(), VecTy, CostKind);
+  SmallVector<const Value *, 2> Args{I->getOperand(0), I->getOperand(1)};
+  return TTI.getArithmeticInstrCost(
+      I->getOpcode(), VecTy, CostKind, TTI::getOperandInfo(I->getOperand(0)),
+      TTI::getOperandInfo(I->getOperand(1)), Args, I);
 }
 
 /// Packs a type's kind and scalar width into one key, so an opcode/intrinsic
@@ -10322,9 +10325,21 @@ bool BoUpSLP::canBuildSplitNode(ArrayRef<Value *> VL,
       (LocalState.getMainOp()->isCast() && LocalState.getAltOp()->isCast()) ||
       (LocalState.getMainOp()->isUnaryOp() &&
        LocalState.getAltOp()->isUnaryOp())) {
+    SmallVector<Value *> Ops0, Ops1;
+    for (Value *V : VL) {
+      auto *I = dyn_cast<Instruction>(V);
+      if (!I)
+        continue;
+      Ops0.push_back(I->getOperand(0));
+      Ops1.push_back(I->getOperand(I->isBinaryOp() ? 1 : 0));
+    }
+    TTI::OperandValueInfo Op1Info = getOperandInfo(Ops0);
+    TTI::OperandValueInfo Op2Info = getOperandInfo(Ops1);
     InstructionCost OriginalVecOpsCost =
-        TTI->getArithmeticInstrCost(Opcode0, VecTy, CostKind) +
-        TTI->getArithmeticInstrCost(Opcode1, VecTy, CostKind);
+        TTI->getArithmeticInstrCost(Opcode0, VecTy, CostKind, Op1Info, Op2Info,
+                                    {}, LocalState.getMainOp()) +
+        TTI->getArithmeticInstrCost(Opcode1, VecTy, CostKind, Op1Info, Op2Info,
+                                    {}, LocalState.getAltOp());
     SmallVector<int> OriginalMask(VL.size(), PoisonMaskElem);
     for (unsigned Idx : seq<unsigned>(VL.size())) {
       if (isa<PoisonValue>(VL[Idx]))
@@ -10335,8 +10350,10 @@ bool BoUpSLP::canBuildSplitNode(ArrayRef<Value *> VL,
         OriginalVecOpsCost + getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy,
                                             CostKind, OriginalMask);
     InstructionCost NewVecOpsCost =
-        TTI->getArithmeticInstrCost(Opcode0, Op1VecTy, CostKind) +
-        TTI->getArithmeticInstrCost(Opcode1, Op2VecTy, CostKind);
+        TTI->getArithmeticInstrCost(Opcode0, Op1VecTy, CostKind, Op1Info,
+                                    Op2Info, {}, LocalState.getMainOp()) +
+        TTI->getArithmeticInstrCost(Opcode1, Op2VecTy, CostKind, Op1Info,
+                                    Op2Info, {}, LocalState.getAltOp());
     InstructionCost NewCost =
         NewVecOpsCost + InsertCost +
         (!VectorizableTree.empty() && getRootNode().hasState() &&
@@ -11138,7 +11155,11 @@ class InstructionsCompatibilityAnalysis {
       case Instruction::FMul:
       case Instruction::FSub:
       case Instruction::FDiv:
-        VectorCost = TTI.getArithmeticInstrCost(MainOpcode, VecTy, Kind);
+        VectorCost = TTI.getArithmeticInstrCost(
+            MainOpcode, VecTy, Kind,
+            TTI::commonOperandInfo(Operands[0][0], Operands[0][1]),
+            TTI::commonOperandInfo(Operands[1][0], Operands[1][1]), {},
+            S.getMainOp());
         break;
       default:
         // Calls (min/max, fmuladd) return above before reaching this switch.
@@ -13688,12 +13709,13 @@ bool BoUpSLP::matchesShlZExt(const TreeEntry &TE, OrdersType &Order,
       getCastContextHint(*getOperandEntry(LhsTE, /*Idx=*/0));
   InstructionCost VecCost =
       TTI->getArithmeticReductionCost(Instruction::Or, VecTy, FMF, CostKind) +
-      TTI->getArithmeticInstrCost(Instruction::Shl, VecTy, CostKind,
-                                  getOperandInfo(LhsTE->Scalars)) +
+      TTI->getArithmeticInstrCost(
+          Instruction::Shl, VecTy, CostKind, getOperandInfo(LhsTE->Scalars),
+          getOperandInfo(RhsTE->Scalars), {}, TE.getMainOp()) +
       TTI->getCastInstrCost(
           Instruction::ZExt, VecTy,
           getWidenedType(SrcScalarTy, LhsTE->getVectorFactor()), CastCtx,
-          CostKind);
+          CostKind, LhsTE->getMainOp());
   InstructionCost BitcastCost = TTI->getCastInstrCost(
       Instruction::BitCast, SrcType, SrcVecTy, CastCtx, CostKind);
   if (!Order.empty()) {
@@ -13726,12 +13748,14 @@ bool BoUpSLP::matchesShlZExt(const TreeEntry &TE, OrdersType &Order,
         IntrinsicCostAttributes CostAttrs(Intrinsic::bswap, SrcType, {SrcType});
         InstructionCost BSwapCost =
             TTI->getMemoryOpCost(Instruction::Load, SrcType, LI->getAlign(),
-                                 LI->getPointerAddressSpace(), CostKind) +
+                                 LI->getPointerAddressSpace(), CostKind,
+                                 TTI::getOperandInfo(LI->getPointerOperand())) +
             TTI->getIntrinsicInstrCost(CostAttrs, CostKind);
         if (BSwapCost <= BitcastCost) {
-          VecCost +=
-              TTI->getMemoryOpCost(Instruction::Load, SrcVecTy, LI->getAlign(),
-                                   LI->getPointerAddressSpace(), CostKind);
+          VecCost += TTI->getMemoryOpCost(
+              Instruction::Load, SrcVecTy, LI->getAlign(),
+              LI->getPointerAddressSpace(), CostKind,
+              TTI::getOperandInfo(LI->getPointerOperand()));
           BitcastCost = BSwapCost;
           ForLoads = true;
         }
@@ -13747,16 +13771,18 @@ bool BoUpSLP::matchesShlZExt(const TreeEntry &TE, OrdersType &Order,
       auto *LI = cast<LoadInst>(SrcTE->getMainOp());
       BitcastCost =
           TTI->getMemoryOpCost(Instruction::Load, SrcType, LI->getAlign(),
-                               LI->getPointerAddressSpace(), CostKind);
+                               LI->getPointerAddressSpace(), CostKind,
+                               TTI::getOperandInfo(LI->getPointerOperand()));
       VecCost +=
           TTI->getMemoryOpCost(Instruction::Load, SrcVecTy, LI->getAlign(),
-                               LI->getPointerAddressSpace(), CostKind);
+                               LI->getPointerAddressSpace(), CostKind,
+                               TTI::getOperandInfo(LI->getPointerOperand()));
       ForLoads = true;
     }
   }
   if (SrcType != ScalarTy) {
     BitcastCost += TTI->getCastInstrCost(Instruction::ZExt, ScalarTy, SrcType,
-                                         TTI::CastContextHint::None, CostKind);
+                                         CastCtx, CostKind);
   }
   return BitcastCost < VecCost;
 }
@@ -13941,10 +13967,10 @@ bool BoUpSLP::matchesInversedZExtSelect(
       getWidenedType(Cmp->getOperand(0)->getType(), CmpTE->getVectorFactor());
   Type *CmpTy = CmpInst::makeCmpResultType(VecTy);
 
-  InstructionCost VecCost =
-      TTI->getCmpSelInstrCost(CmpTE->getOpcode(), VecTy, CmpTy, MainPred,
-                              CostKind, getOperandInfo(CmpTE->getOperand(0)),
-                              getOperandInfo(CmpTE->getOperand(1)));
+  InstructionCost VecCost = TTI->getCmpSelInstrCost(
+      CmpTE->getOpcode(), VecTy, CmpTy, MainPred, CostKind,
+      getOperandInfo(CmpTE->getOperand(0)),
+      getOperandInfo(CmpTE->getOperand(1)), CmpTE->getMainOp());
   InstructionCost BVCost = getScalarizationOverhead(
       *TTI, SLPReVec, Cmp->getType(), cast<VectorType>(CmpTy),
       APInt::getAllOnes(CmpTE->getVectorFactor()),
@@ -14014,12 +14040,25 @@ bool BoUpSLP::matchesSelectOfBits(const TreeEntry &SelectTE) const {
     BitcastCost += TTI->getCastInstrCost(Instruction::ZExt, ScalarTy, DstTy,
                                          TTI::CastContextHint::None, CostKind);
   }
+  // Use the condition predicate shared by all lanes, if there is one.
+  CmpPredicate SelPred = CmpInst::BAD_ICMP_PREDICATE;
+  CmpPredicate P;
+  if (match(SelectTE.getMainOp(),
+            m_Select(m_ICmp(P, m_Value(), m_Value()), m_Value(), m_Value())) &&
+      all_of(SelectTE.Scalars, [&](Value *V) {
+        CmpPredicate LaneP;
+        return match(V, m_Select(m_ICmp(LaneP, m_Value(), m_Value()), m_Value(),
+                                 m_Value())) &&
+               LaneP == P.dropSameSign();
+      }))
+    SelPred = P;
+
   FastMathFlags FMF;
   InstructionCost SelectCost =
-      TTI->getCmpSelInstrCost(Instruction::Select, VecTy, CmpTy,
-                              CmpInst::BAD_ICMP_PREDICATE, CostKind,
-                              getOperandInfo(Op1TE->Scalars),
-                              getOperandInfo(Op2TE->Scalars)) +
+      TTI->getCmpSelInstrCost(Instruction::Select, VecTy, CmpTy, SelPred,
+                              CostKind, getOperandInfo(Op1TE->Scalars),
+                              getOperandInfo(Op2TE->Scalars),
+                              SelectTE.getMainOp()) +
       TTI->getArithmeticReductionCost(Instruction::Or, VecTy, FMF, CostKind);
   return BitcastCost <= SelectCost;
 }
@@ -14330,9 +14369,10 @@ void BoUpSLP::transformNodes() {
         inversePermutation(E.ReorderIndices, Mask);
         auto *BaseLI = cast<LoadInst>(E.Scalars.back());
         InstructionCost OriginalVecCost =
-            TTI->getMemoryOpCost(Instruction::Load, VecTy, BaseLI->getAlign(),
-                                 BaseLI->getPointerAddressSpace(), CostKind,
-                                 TTI::OperandValueInfo()) +
+            TTI->getMemoryOpCost(
+                Instruction::Load, VecTy, BaseLI->getAlign(),
+                BaseLI->getPointerAddressSpace(), CostKind,
+                TTI::getOperandInfo(BaseLI->getPointerOperand())) +
             getShuffleCost(*TTI, TTI::SK_Reverse, VecTy, CostKind, Mask);
         InstructionCost StridedCost = TTI->getMemIntrinsicInstrCost(
             MemIntrinsicCostAttributes(Intrinsic::experimental_vp_strided_load,
@@ -14373,7 +14413,7 @@ void BoUpSLP::transformNodes() {
         InstructionCost OriginalVecCost =
             TTI->getMemoryOpCost(Instruction::Store, VecTy, BaseSI->getAlign(),
                                  BaseSI->getPointerAddressSpace(), CostKind,
-                                 TTI::OperandValueInfo()) +
+                                 getOperandInfo(E.getOperand(0))) +
             getShuffleCost(*TTI, TTI::SK_Reverse, VecTy, CostKind, Mask);
         InstructionCost StridedCost = TTI->getMemIntrinsicInstrCost(
             MemIntrinsicCostAttributes(Intrinsic::experimental_vp_strided_store,
@@ -15043,7 +15083,7 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
           CastOpcode = IsSigned ? Instruction::SExt : Instruction::ZExt;
         return TTI.getCastInstrCost(CastOpcode, getWidenedType(ScalarTy, VF),
                                     getWidenedType(EScalarTy, VF),
-                                    TTI::CastContextHint::None, CostKind);
+                                    R.getCastContextHint(E), CostKind);
       }
       return TTI::TCC_Free;
     };
@@ -15059,9 +15099,12 @@ class BoUpSLP::ShuffleCostEstimator : public BaseShuffleAnalysis {
         unsigned SrcSz = R.DL->getTypeSizeInBits(EScalarTy);
         if (DstSz > SrcSz)
           CastOpcode = IsSigned ? Instruction::SExt : Instruction::ZExt;
+        TTI::CastContextHint CCH = TTI::CastContextHint::None;
+        if (ArrayRef<TreeEntry *> TEs = R.getTreeEntries(V); TEs.size() == 1)
+          CCH = R.getCastContextHint(*TEs.front());
         return TTI.getCastInstrCost(
             CastOpcode, VectorType::get(ScalarTy, VecTy->getElementCount()),
-            VecTy, TTI::CastContextHint::None, CostKind);
+            VecTy, CCH, CostKind);
       }
       return TTI::TCC_Free;
     };
@@ -16212,7 +16255,7 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
           TTI::OperandValueInfo Op1Info = TTI::getOperandInfo(I->getOperand(0));
           TTI::OperandValueInfo Op2Info = TTI::getOperandInfo(I->getOperand(1));
           PeeledScalarCost += TTI->getArithmeticInstrCost(
-              I->getOpcode(), OrigScalarTy, CostKind, Op1Info, Op2Info);
+              I->getOpcode(), OrigScalarTy, CostKind, Op1Info, Op2Info, {}, I);
         }
         bool PeeledCostAdded = false;
         InstructionCost CostDiff = GetCostDiff(
@@ -16273,8 +16316,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
       auto *CI = cast<CmpInst>(VI->getOperand(0));
       IntrinsicCost -= TTI->getCmpSelInstrCost(
           CI->getOpcode(), Ty, Builder.getInt1Ty(), CI->getPredicate(),
-          CostKind, {TTI::OK_AnyValue, TTI::OP_None},
-          {TTI::OK_AnyValue, TTI::OP_None}, CI);
+          CostKind, TTI::getOperandInfo(CI->getOperand(0)),
+          TTI::getOperandInfo(CI->getOperand(1)), CI);
     }
     return IntrinsicCost;
   };
@@ -16505,6 +16548,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
       // vector and the original aggregate.
       Align VecAlign = std::max(DL->getPrefTypeAlign(SrcVecTy),
                                 DL->getPrefTypeAlign(VL0->getType()));
+      // The stack slot roundtrip is synthesized by codegen; there is no IR
+      // value or pointer to take operand info from.
       Cost += TTI->getMemoryOpCost(Instruction::Store, SrcVecTy, VecAlign,
                                    /*AddressSpace=*/0, CostKind) +
               TTI->getMemoryOpCost(Instruction::Load, VL0->getType(), VecAlign,
@@ -16805,9 +16850,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
             TTI.getIntrinsicInstrCost(CostAttrs, CostKind);
         BitcastCost += IntrinsicCost;
         if (SrcType != ScalarTy) {
-          BitcastCost +=
-              TTI.getCastInstrCost(Instruction::ZExt, ScalarTy, SrcType,
-                                   TTI::CastContextHint::None, CostKind);
+          BitcastCost += TTI.getCastInstrCost(Instruction::ZExt, ScalarTy,
+                                              SrcType, CastCtx, CostKind);
         }
       }
       return BitcastCost + CommonCost;
@@ -16851,7 +16895,7 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
         if (SrcType != ScalarTy) {
           LoadCost +=
               TTI.getCastInstrCost(Instruction::ZExt, ScalarTy, SrcType,
-                                   TTI::CastContextHint::None, CostKind);
+                                   getCastContextHint(*LoadTE), CostKind);
         }
       }
       return LoadCost + CommonCost;
@@ -16966,10 +17010,11 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
       }
       TTI::OperandValueInfo Op1Info = TTI::getOperandInfo(Op1);
       TTI::OperandValueInfo Op2Info = TTI::getOperandInfo(Op2);
+      auto *I = dyn_cast<Instruction>(UniqueValues[Idx]);
       InstructionCost ScalarCost = TTI->getArithmeticInstrCost(
-          ShuffleOrOp, OrigScalarTy, CostKind, Op1Info, Op2Info, Operands);
-      if (auto *I = dyn_cast<Instruction>(UniqueValues[Idx]);
-          I && (ShuffleOrOp == Instruction::FAdd ||
+          ShuffleOrOp, OrigScalarTy, CostKind, Op1Info, Op2Info, Operands,
+          I && I->getOpcode() == ShuffleOrOp ? I : nullptr);
+      if (I && (ShuffleOrOp == Instruction::FAdd ||
                 ShuffleOrOp == Instruction::FSub)) {
         InstructionCost IntrinsicCost = GetFMulAddCost(E->getOperations(), I);
         if (IntrinsicCost.isValid())
@@ -17000,7 +17045,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
       TTI::OperandValueInfo Op1Info = getOperandInfo(E->getOperand(0));
       TTI::OperandValueInfo Op2Info = getOperandInfo(E->getOperand(OpIdx));
       InstructionCost Cost = TTI->getArithmeticInstrCost(
-          ShuffleOrOp, VecTy, CostKind, Op1Info, Op2Info, {}, nullptr, TLI);
+          ShuffleOrOp, VecTy, CostKind, Op1Info, Op2Info, {},
+          VL0->getOpcode() == ShuffleOrOp ? VL0 : nullptr, TLI);
       // N columns need N-1 vector combines; price extra columns
       // conservatively, skipping identity-only columns (not combined by
       // codegen).
@@ -17011,9 +17057,12 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
                 return isBinOpIdentityConstant(V, CombineOpcode);
               }))
             continue;
+          // The first operand is the running fold, a computed value with no
+          // static properties to model.
           Cost += TTI->getArithmeticInstrCost(
               ShuffleOrOp, VecTy, CostKind, {},
-              getOperandInfo(E->getOperand(Idx)), {}, nullptr, TLI);
+              getOperandInfo(E->getOperand(Idx)), {},
+              VL0->getOpcode() == ShuffleOrOp ? VL0 : nullptr, TLI);
         }
       }
       return Cost + CommonCost;
@@ -17026,9 +17075,10 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
   case Instruction::Load: {
     auto GetScalarCost = [&](unsigned Idx) {
       auto *VI = cast<LoadInst>(UniqueValues[Idx]);
-      return TTI->getMemoryOpCost(Instruction::Load, OrigScalarTy,
-                                  VI->getAlign(), VI->getPointerAddressSpace(),
-                                  CostKind, TTI::OperandValueInfo(), VI);
+      return TTI->getMemoryOpCost(
+          Instruction::Load, OrigScalarTy, VI->getAlign(),
+          VI->getPointerAddressSpace(), CostKind,
+          TTI::getOperandInfo(VI->getPointerOperand()), VI);
     };
     auto *LI0 = cast<LoadInst>(VL0);
     auto GetVectorCost = [&](InstructionCost CommonCost) {
@@ -17043,7 +17093,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
         } else {
           VecLdCost = TTI->getMemoryOpCost(
               Instruction::Load, VecTy, LI0->getAlign(),
-              LI0->getPointerAddressSpace(), CostKind, TTI::OperandValueInfo());
+              LI0->getPointerAddressSpace(), CostKind,
+              TTI::getOperandInfo(LI0->getPointerOperand()));
         }
         break;
       case TreeEntry::StridedVectorize: {
@@ -17101,7 +17152,8 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
         } else {
           VecLdCost = TTI->getMemoryOpCost(
               Instruction::Load, LoadVecTy, CommonAlignment,
-              LI0->getPointerAddressSpace(), CostKind, TTI::OperandValueInfo());
+              LI0->getPointerAddressSpace(), CostKind,
+              TTI::getOperandInfo(LI0->getPointerOperand()));
           // TODO: include this cost into CommonCost.
           VecLdCost += getShuffleCost(*TTI, TTI::SK_PermuteSingleSrc, LoadVecTy,
                                       CostKind, CompressMask);
@@ -17311,9 +17363,10 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
           for (unsigned Idx : seq<unsigned>(1, E->getNumOperands())) {
             TTI::OperandValueInfo ColInfo = getOperandInfo(E->getOperand(Idx));
             if (!RunningInfo.isConstant() || !ColInfo.isConstant())
-              Cost += TTIRef.getArithmeticInstrCost(Opcode, VecTy, CostKind,
-                                                    RunningInfo, ColInfo, {},
-                                                    nullptr, TLI);
+              Cost += TTIRef.getArithmeticInstrCost(
+                  Opcode, VecTy, CostKind, RunningInfo, ColInfo, {},
+                  Opcode == E->getOpcode() ? E->getMainOp() : E->getAltOp(),
+                  TLI);
             TTI::OperandValueKind Kind = TTI::OK_AnyValue;
             if (RunningInfo.isConstant() && ColInfo.isConstant())
               Kind = RunningInfo.Kind == TTI::OK_UniformConstantValue &&
@@ -17329,15 +17382,15 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
         VecCost = ChainCost(E->getOpcode()) + ChainCost(E->getAltOpcode());
       } else if (auto *CI0 = dyn_cast<CmpInst>(VL0)) {
         auto *MaskTy = getWidenedType(Builder.getInt1Ty(), VL.size());
-        VecCost = TTIRef.getCmpSelInstrCost(
-            E->getOpcode(), VecTy, MaskTy, CI0->getPredicate(), CostKind,
-            {TTI::OK_AnyValue, TTI::OP_None}, {TTI::OK_AnyValue, TTI::OP_None},
-            VL0);
+        TTI::OperandValueInfo Op1Info = getOperandInfo(E->getOperand(0));
+        TTI::OperandValueInfo Op2Info = getOperandInfo(E->getOperand(1));
+        VecCost = TTIRef.getCmpSelInstrCost(E->getOpcode(), VecTy, MaskTy,
+                                            CI0->getPredicate(), CostKind,
+                                            Op1Info, Op2Info, VL0);
         VecCost += TTIRef.getCmpSelInstrCost(
             E->getOpcode(), VecTy, MaskTy,
-            cast<CmpInst>(E->getAltOp())->getPredicate(), CostKind,
-            {TTI::OK_AnyValue, TTI::OP_None}, {TTI::OK_AnyValue, TTI::OP_None},
-            E->getAltOp());
+            cast<CmpInst>(E->getAltOp())->getPredicate(), CostKind, Op1Info,
+            Op2Info, E->getAltOp());
       } else {
         Type *SrcSclTy = E->getMainOp()->getOperand(0)->getType();
         auto *SrcTy = getWidenedType(SrcSclTy, VL.size());
@@ -17353,9 +17406,9 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
           }
           if (BWSz <= SrcBWSz) {
             if (BWSz < SrcBWSz)
-              VecCost =
-                  TTIRef.getCastInstrCost(Instruction::Trunc, VecTy, SrcTy,
-                                          TTI::CastContextHint::None, CostKind);
+              VecCost = TTIRef.getCastInstrCost(
+                  Instruction::Trunc, VecTy, SrcTy,
+                  GetCastContextHint(E->getMainOp()->getOperand(0)), CostKind);
             LLVM_DEBUG({
               dbgs()
                   << "SLP: alternate extension, which should be truncated.\n";
@@ -17364,11 +17417,14 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
             return VecCost;
           }
         }
-        VecCost = TTIRef.getCastInstrCost(E->getOpcode(), VecTy, SrcTy,
-                                          TTI::CastContextHint::None, CostKind);
-        VecCost +=
-            TTIRef.getCastInstrCost(E->getAltOpcode(), VecTy, SrcTy,
-                                    TTI::CastContextHint::None, CostKind);
+        VecCost = TTIRef.getCastInstrCost(
+            E->getOpcode(), VecTy, SrcTy,
+            GetCastContextHint(E->getMainOp()->getOperand(0)), CostKind,
+            E->getMainOp());
+        VecCost += TTIRef.getCastInstrCost(
+            E->getAltOpcode(), VecTy, SrcTy,
+            GetCastContextHint(E->getMainOp()->getOperand(0)), CostKind,
+            E->getAltOp());
       }
       SmallVector<int> Mask;
       E->buildAltOpShuffleMask(
@@ -19842,7 +19898,7 @@ InstructionCost BoUpSLP::getTreeCost(InstructionCost TreeCost,
                   VecOpcode, FTy,
                   getWidenedType(IntegerType::get(FTy->getContext(), BWSz),
                                  FTy->getNumElements()),
-                  TTI::CastContextHint::None, CostKind);
+                  getCastContextHint(*ScalarTE), CostKind);
               LLVM_DEBUG(dbgs() << "SLP: Adding cost " << C
                                 << " for extending externally used vector with "
                                    "non-equal minimum bitwidth.\n");
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp
index 382adb71cede2..8279c277a1f04 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer/SLPMemoryUtils.cpp
@@ -297,7 +297,8 @@ bool isMaskedLoadCompress(
   } else {
     LoadCost =
         TTI.getMemoryOpCost(Instruction::Load, LoadVecTy, CommonAlignment,
-                            LI->getPointerAddressSpace(), CostKind);
+                            LI->getPointerAddressSpace(), CostKind,
+                            TTI::getOperandInfo(LI->getPointerOperand()));
   }
   if (IsStrided && !IsMasked && Order.empty()) {
     // Check for potential segmented(interleaved) loads.
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
index c98a4d1f2c182..ab1f6ce5aa544 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/externally-used-copyables.ll
@@ -4,7 +4,7 @@
 define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i64 %.sroa.3308.0.copyload, i64 %.neg1, i64 %indvar3788, i64 %4, i64 %5, i64 %6, i64 %7, i64 %8) {
 ; CHECK-LABEL: define void @test(
 ; CHECK-SAME: i64 [[TMP0:%.*]], i64 [[TMP1:%.*]], i64 [[TMP2:%.*]], i64 [[TMP3:%.*]], i64 [[DOTSROA_3341_0_COPYLOAD:%.*]], i64 [[DOTSROA_3308_0_COPYLOAD:%.*]], i64 [[DOTNEG1:%.*]], i64 [[INDVAR3788:%.*]], i64 [[TMP4:%.*]], i64 [[TMP5:%.*]], i64 [[TMP6:%.*]], i64 [[TMP7:%.*]], i64 [[TMP8:%.*]]) #[[ATTR0:[0-9]+]] {
-; CHECK-NEXT:  [[_LR_PH_PREHEADER:.*:]]
+; CHECK-NEXT:  [[_LR_PH_PREHEADER:.*]]:
 ; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <2 x i64> poison, i64 [[TMP1]], i64 0
 ; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <2 x i64> [[TMP9]], i64 [[TMP0]], i64 1
 ; CHECK-NEXT:    [[TMP13:%.*]] = mul <2 x i64> [[TMP10]], <i64 1, i64 24>
@@ -14,10 +14,9 @@ define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i6
 ; CHECK-NEXT:    [[TMP15:%.*]] = shufflevector <2 x i64> [[TMP14]], <2 x i64> <i64 -1, i64 poison>, <2 x i32> <i32 2, i32 0>
 ; CHECK-NEXT:    [[TMP16:%.*]] = sub <2 x i64> [[TMP14]], [[TMP15]]
 ; CHECK-NEXT:    [[TMP20:%.*]] = shl i64 [[TMP0]], 11
-; CHECK-NEXT:    [[TMP35:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <2 x i64> [[TMP35]], <2 x i64> poison, <2 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP25:%.*]] = shl <2 x i64> [[TMP12]], <i64 11, i64 0>
 ; CHECK-NEXT:    [[TMP21:%.*]] = sub i64 1, [[TMP20]]
+; CHECK-NEXT:    [[TMP28:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 1
+; CHECK-NEXT:    [[TMP25:%.*]] = insertelement <2 x i64> [[TMP28]], i64 [[TMP20]], i64 0
 ; CHECK-NEXT:    [[TMP22:%.*]] = add <2 x i64> [[TMP25]], <i64 8, i64 1>
 ; CHECK-NEXT:    [[TMP23:%.*]] = or <2 x i64> [[TMP25]], <i64 8, i64 1>
 ; CHECK-NEXT:    [[TMP33:%.*]] = shufflevector <2 x i64> [[TMP22]], <2 x i64> [[TMP23]], <2 x i32> <i32 0, i32 3>
@@ -37,6 +36,8 @@ define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i6
 ; CHECK-NEXT:    [[TMP38:%.*]] = insertelement <4 x i64> [[TMP37]], i64 [[TMP8]], i64 1
 ; CHECK-NEXT:    [[TMP39:%.*]] = insertelement <4 x i64> [[TMP38]], i64 [[DOTSROA_3308_0_COPYLOAD]], i64 2
 ; CHECK-NEXT:    [[TMP40:%.*]] = insertelement <4 x i64> [[TMP39]], i64 [[TMP0]], i64 3
+; CHECK-NEXT:    [[TMP64:%.*]] = insertelement <2 x i64> poison, i64 [[TMP0]], i64 0
+; CHECK-NEXT:    [[TMP12:%.*]] = shufflevector <2 x i64> [[TMP64]], <2 x i64> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP41:%.*]] = shufflevector <2 x i64> [[TMP17]], <2 x i64> poison, <8 x i32> <i32 0, i32 1, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison, i32 poison>
 ; CHECK-NEXT:    [[TMP63:%.*]] = insertelement <4 x i64> poison, i64 [[TMP0]], i64 1
 ; CHECK-NEXT:    [[TMP43:%.*]] = mul i64 [[TMP0]], [[TMP0]]
@@ -46,8 +47,8 @@ define void @test(i64 %0, i64 %1, i64 %2, i64 %3, i64 %.sroa.3341.0.copyload, i6
 ; CHECK-NEXT:    [[TMP47:%.*]] = shufflevector <6 x i64> [[TMP31]], <6 x i64> [[TMP46]], <6 x i32> <i32 0, i32 1, i32 2, i32 3, i32 6, i32 7>
 ; CHECK-NEXT:    [[TMP48:%.*]] = shufflevector <2 x i64> [[TMP16]], <2 x i64> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
 ; CHECK-NEXT:    br label %[[DOTLR_PH1977_US:.*]]
-; CHECK:       [[_LR_PH1977_US:.*:]]
-; CHECK-NEXT:    [[INDVAR37888:%.*]] = phi i64 [ 0, [[DOTLR_PH_PREHEADER:%.*]] ], [ 1, %[[DOTLR_PH1977_US]] ]
+; CHECK:       [[DOTLR_PH1977_US]]:
+; CHECK-NEXT:    [[INDVAR37888:%.*]] = phi i64 [ 0, %[[_LR_PH_PREHEADER]] ], [ 1, %[[DOTLR_PH1977_US]] ]
 ; CHECK-NEXT:    [[TMP49:%.*]] = shufflevector <6 x i64> [[TMP47]], <6 x i64> poison, <8 x i32> <i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 5, i32 5>
 ; CHECK-NEXT:    [[TMP50:%.*]] = mul <8 x i64> [[TMP30]], [[TMP49]]
 ; CHECK-NEXT:    [[TMP51:%.*]] = or <8 x i64> [[TMP30]], [[TMP49]]
diff --git a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
index c4c97622ac726..05e5e5f29d51b 100644
--- a/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AMDGPU/elementwise-fma-operand1.ll
@@ -8,7 +8,8 @@
 
 ; Elementwise d = c + a * b, where the fmul is operand 1 of the fadd. These
 ; targets halve the cost of a packed fmul, so SLP is tempted to vectorize and
-; break the scalar fma chain. The 14 runs sit at the cost boundary. The 12 runs
+; break the scalar fma chain. The scalar fmul is costed as free once the fma
+; fusion context is visible, so the 14 runs stay scalar. The 12 runs
 ; vectorize either way and guard against the fmuladd marking landing on the load
 ; at operand 0 after the fma detection picked the fmul at operand 1, which
 ; asserts. axpy4_mixed_reassoc carries reassoc on one lane only, so the whole
@@ -18,12 +19,42 @@ define void @axpy4_contract(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr
 ; CHECK-LABEL: define void @axpy4_contract(
 ; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = fmul contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP4:%.*]] = fadd contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT:    store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT:    [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT:    [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT:    [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT:    [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT:    [[R0:%.*]] = fadd contract float [[C0]], [[M0]]
+; CHECK-NEXT:    store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT:    [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT:    [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT:    [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT:    [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT:    [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT:    [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT:    [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT:    [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; CHECK-NEXT:    [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT:    store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT:    [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT:    [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT:    [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT:    [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT:    [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT:    [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT:    [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; CHECK-NEXT:    [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; CHECK-NEXT:    [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT:    store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT:    [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT:    [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT:    [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT:    [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT:    [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT:    [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT:    [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; CHECK-NEXT:    [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; CHECK-NEXT:    [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT:    store float [[R3]], ptr [[DP3]], align 4
 ; CHECK-NEXT:    ret void
 ;
 ; THR12-LABEL: define void @axpy4_contract(
@@ -81,12 +112,42 @@ define void @axpy4_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b, ptr n
 ; CHECK-LABEL: define void @axpy4_reassoc(
 ; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = fmul reassoc contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP4:%.*]] = fadd reassoc contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT:    store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT:    [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT:    [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT:    [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT:    [[M0:%.*]] = fmul reassoc contract float [[A0]], [[B0]]
+; CHECK-NEXT:    [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; CHECK-NEXT:    store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT:    [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT:    [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT:    [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT:    [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT:    [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT:    [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT:    [[M1:%.*]] = fmul reassoc contract float [[A1]], [[B1]]
+; CHECK-NEXT:    [[R1:%.*]] = fadd reassoc contract float [[C1]], [[M1]]
+; CHECK-NEXT:    [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT:    store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT:    [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT:    [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT:    [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT:    [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT:    [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT:    [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT:    [[M2:%.*]] = fmul reassoc contract float [[A2]], [[B2]]
+; CHECK-NEXT:    [[R2:%.*]] = fadd reassoc contract float [[C2]], [[M2]]
+; CHECK-NEXT:    [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT:    store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT:    [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT:    [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT:    [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT:    [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT:    [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT:    [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT:    [[M3:%.*]] = fmul reassoc contract float [[A3]], [[B3]]
+; CHECK-NEXT:    [[R3:%.*]] = fadd reassoc contract float [[C3]], [[M3]]
+; CHECK-NEXT:    [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT:    store float [[R3]], ptr [[DP3]], align 4
 ; CHECK-NEXT:    ret void
 ;
 ; THR12-LABEL: define void @axpy4_reassoc(
@@ -144,12 +205,42 @@ define void @axpy4_mixed_reassoc(ptr noalias %d, ptr noalias %a, ptr noalias %b,
 ; CHECK-LABEL: define void @axpy4_mixed_reassoc(
 ; CHECK-SAME: ptr noalias [[D:%.*]], ptr noalias [[A:%.*]], ptr noalias [[B:%.*]], ptr noalias [[C:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x float>, ptr [[C]], align 4
-; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x float>, ptr [[A]], align 4
-; CHECK-NEXT:    [[TMP2:%.*]] = load <4 x float>, ptr [[B]], align 4
-; CHECK-NEXT:    [[TMP3:%.*]] = fmul contract <4 x float> [[TMP1]], [[TMP2]]
-; CHECK-NEXT:    [[TMP4:%.*]] = fadd contract <4 x float> [[TMP0]], [[TMP3]]
-; CHECK-NEXT:    store <4 x float> [[TMP4]], ptr [[D]], align 4
+; CHECK-NEXT:    [[C0:%.*]] = load float, ptr [[C]], align 4
+; CHECK-NEXT:    [[A0:%.*]] = load float, ptr [[A]], align 4
+; CHECK-NEXT:    [[B0:%.*]] = load float, ptr [[B]], align 4
+; CHECK-NEXT:    [[M0:%.*]] = fmul contract float [[A0]], [[B0]]
+; CHECK-NEXT:    [[R0:%.*]] = fadd reassoc contract float [[C0]], [[M0]]
+; CHECK-NEXT:    store float [[R0]], ptr [[D]], align 4
+; CHECK-NEXT:    [[CP1:%.*]] = getelementptr inbounds float, ptr [[C]], i64 1
+; CHECK-NEXT:    [[C1:%.*]] = load float, ptr [[CP1]], align 4
+; CHECK-NEXT:    [[AP1:%.*]] = getelementptr inbounds float, ptr [[A]], i64 1
+; CHECK-NEXT:    [[A1:%.*]] = load float, ptr [[AP1]], align 4
+; CHECK-NEXT:    [[BP1:%.*]] = getelementptr inbounds float, ptr [[B]], i64 1
+; CHECK-NEXT:    [[B1:%.*]] = load float, ptr [[BP1]], align 4
+; CHECK-NEXT:    [[M1:%.*]] = fmul contract float [[A1]], [[B1]]
+; CHECK-NEXT:    [[R1:%.*]] = fadd contract float [[C1]], [[M1]]
+; CHECK-NEXT:    [[DP1:%.*]] = getelementptr inbounds float, ptr [[D]], i64 1
+; CHECK-NEXT:    store float [[R1]], ptr [[DP1]], align 4
+; CHECK-NEXT:    [[CP2:%.*]] = getelementptr inbounds float, ptr [[C]], i64 2
+; CHECK-NEXT:    [[C2:%.*]] = load float, ptr [[CP2]], align 4
+; CHECK-NEXT:    [[AP2:%.*]] = getelementptr inbounds float, ptr [[A]], i64 2
+; CHECK-NEXT:    [[A2:%.*]] = load float, ptr [[AP2]], align 4
+; CHECK-NEXT:    [[BP2:%.*]] = getelementptr inbounds float, ptr [[B]], i64 2
+; CHECK-NEXT:    [[B2:%.*]] = load float, ptr [[BP2]], align 4
+; CHECK-NEXT:    [[M2:%.*]] = fmul contract float [[A2]], [[B2]]
+; CHECK-NEXT:    [[R2:%.*]] = fadd contract float [[C2]], [[M2]]
+; CHECK-NEXT:    [[DP2:%.*]] = getelementptr inbounds float, ptr [[D]], i64 2
+; CHECK-NEXT:    store float [[R2]], ptr [[DP2]], align 4
+; CHECK-NEXT:    [[CP3:%.*]] = getelementptr inbounds float, ptr [[C]], i64 3
+; CHECK-NEXT:    [[C3:%.*]] = load float, ptr [[CP3]], align 4
+; CHECK-NEXT:    [[AP3:%.*]] = getelementptr inbounds float, ptr [[A]], i64 3
+; CHECK-NEXT:    [[A3:%.*]] = load float, ptr [[AP3]], align 4
+; CHECK-NEXT:    [[BP3:%.*]] = getelementptr inbounds float, ptr [[B]], i64 3
+; CHECK-NEXT:    [[B3:%.*]] = load float, ptr [[BP3]], align 4
+; CHECK-NEXT:    [[M3:%.*]] = fmul contract float [[A3]], [[B3]]
+; CHECK-NEXT:    [[R3:%.*]] = fadd contract float [[C3]], [[M3]]
+; CHECK-NEXT:    [[DP3:%.*]] = getelementptr inbounds float, ptr [[D]], i64 3
+; CHECK-NEXT:    store float [[R3]], ptr [[DP3]], align 4
 ; CHECK-NEXT:    ret void
 ;
 ; THR12-LABEL: define void @axpy4_mixed_reassoc(
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/non-commutative-second-arg-only-copyable.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/non-commutative-second-arg-only-copyable.ll
index 442b86733fe96..dae264f6c96cd 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/non-commutative-second-arg-only-copyable.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/non-commutative-second-arg-only-copyable.ll
@@ -6,11 +6,11 @@ define i32 @main(ptr %q, ptr %a, i8 %.pre) {
 ; CHECK-SAME: ptr [[Q:%.*]], ptr [[A:%.*]], i8 [[DOTPRE:%.*]]) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
 ; CHECK-NEXT:    [[DOTPRE1:%.*]] = load i8, ptr [[Q]], align 1
-; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x i8> poison, i8 [[DOTPRE]], i64 0
-; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x i8> [[TMP0]], i8 [[DOTPRE1]], i64 1
-; CHECK-NEXT:    [[TMP2:%.*]] = sext <2 x i8> [[TMP1]] to <2 x i32>
-; CHECK-NEXT:    [[TMP3:%.*]] = add <2 x i32> [[TMP2]], <i32 0, i32 1>
-; CHECK-NEXT:    [[TMP4:%.*]] = shufflevector <2 x i32> [[TMP2]], <2 x i32> <i32 poison, i32 1>, <2 x i32> <i32 0, i32 3>
+; CHECK-NEXT:    [[CONV11_I:%.*]] = sext i8 [[DOTPRE]] to i32
+; CHECK-NEXT:    [[TMP0:%.*]] = sext i8 [[DOTPRE1]] to i32
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[TMP0]], 1
+; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <2 x i32> <i32 poison, i32 1>, i32 [[CONV11_I]], i64 0
+; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <2 x i32> [[TMP4]], i32 [[TMP1]], i64 1
 ; CHECK-NEXT:    [[TMP5:%.*]] = shl <2 x i32> [[TMP4]], [[TMP3]]
 ; CHECK-NEXT:    [[TMP6:%.*]] = trunc <2 x i32> [[TMP5]] to <2 x i16>
 ; CHECK-NEXT:    store <2 x i16> [[TMP6]], ptr [[A]], align 2
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/reduced-copyable-element.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/reduced-copyable-element.ll
index 7c35c888fe901..066abb629e8ff 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/reduced-copyable-element.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/reduced-copyable-element.ll
@@ -13,8 +13,10 @@ define i32 @main() {
 ; CHECK-NEXT:    [[TMP3:%.*]] = sext <2 x i32> [[TMP2]] to <2 x i64>
 ; CHECK-NEXT:    [[TMP4:%.*]] = call <2 x i64> @llvm.umin.v2i64(<2 x i64> [[TMP3]], <2 x i64> splat (i64 17179869184))
 ; CHECK-NEXT:    [[TMP5:%.*]] = trunc <2 x i64> [[TMP4]] to <2 x i32>
-; CHECK-NEXT:    [[TMP6:%.*]] = add <2 x i32> [[TMP5]], <i32 0, i32 1>
-; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @llvm.vector.reduce.or.v2i32(<2 x i32> [[TMP6]])
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <2 x i32> [[TMP5]], i64 1
+; CHECK-NEXT:    [[TMP9:%.*]] = add i32 [[TMP6]], 1
+; CHECK-NEXT:    [[TMP8:%.*]] = extractelement <2 x i32> [[TMP5]], i64 0
+; CHECK-NEXT:    [[TMP7:%.*]] = or i32 [[TMP9]], [[TMP8]]
 ; CHECK-NEXT:    ret i32 [[TMP7]]
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/RISCV/strided-loads-with-external-indices.ll b/llvm/test/Transforms/SLPVectorizer/RISCV/strided-loads-with-external-indices.ll
index b1d045b22a730..604be4b570998 100644
--- a/llvm/test/Transforms/SLPVectorizer/RISCV/strided-loads-with-external-indices.ll
+++ b/llvm/test/Transforms/SLPVectorizer/RISCV/strided-loads-with-external-indices.ll
@@ -10,10 +10,14 @@ define void @test() {
 ; CHECK-NEXT:    [[SUB4_I_I65_US:%.*]] = or i64 0, 1
 ; CHECK-NEXT:    br label [[BODY:%.*]]
 ; CHECK:       body:
-; CHECK-NEXT:    [[TMP0:%.*]] = call <2 x i32> @llvm.masked.gather.v2i32.v2p0(<2 x ptr> align 4 getelementptr ([[CLASS_A:%.*]], <2 x ptr> splat (ptr null), <2 x i64> <i64 0, i64 1>), <2 x i1> splat (i1 true), <2 x i32> poison)
-; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <2 x i32> [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <2 x i32> [[TMP0]], i64 1
-; CHECK-NEXT:    [[CMP_I_I_I_I67_US:%.*]] = icmp slt i32 [[TMP1]], [[TMP2]]
+; CHECK-NEXT:    [[ADD_I_I62_US:%.*]] = shl i64 0, 0
+; CHECK-NEXT:    [[TMP0:%.*]] = insertelement <2 x i64> <i64 poison, i64 1>, i64 [[ADD_I_I62_US]], i64 0
+; CHECK-NEXT:    [[TMP1:%.*]] = or <2 x i64> zeroinitializer, [[TMP0]]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr [[CLASS_A:%.*]], <2 x ptr> splat (ptr null), <2 x i64> [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = call <2 x i32> @llvm.masked.gather.v2i32.v2p0(<2 x ptr> align 4 [[TMP2]], <2 x i1> splat (i1 true), <2 x i32> poison)
+; CHECK-NEXT:    [[TMP4:%.*]] = extractelement <2 x i32> [[TMP3]], i64 0
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <2 x i32> [[TMP3]], i64 1
+; CHECK-NEXT:    [[CMP_I_I_I_I67_US:%.*]] = icmp slt i32 [[TMP4]], [[TMP5]]
 ; CHECK-NEXT:    [[SPEC_SELECT_I_I68_US:%.*]] = select i1 false, i64 [[SUB4_I_I65_US]], i64 0
 ; CHECK-NEXT:    br label [[BODY]]
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/SystemZ/non-power-2-subvector-extract.ll b/llvm/test/Transforms/SLPVectorizer/SystemZ/non-power-2-subvector-extract.ll
index bde56da998c0b..d6cd8b6b12d27 100644
--- a/llvm/test/Transforms/SLPVectorizer/SystemZ/non-power-2-subvector-extract.ll
+++ b/llvm/test/Transforms/SLPVectorizer/SystemZ/non-power-2-subvector-extract.ll
@@ -8,18 +8,34 @@ define void @p() {
 ; CHECK-LABEL: define void @p(
 ; CHECK-SAME: ) #[[ATTR0:[0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
-; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @c, i64 52), align 4
-; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x i32> [[TMP0]], i64 0
-; CHECK-NEXT:    [[TMP2:%.*]] = extractelement <4 x i32> [[TMP0]], i64 2
-; CHECK-NEXT:    [[TMP3:%.*]] = xor <4 x i32> [[TMP0]], splat (i32 1)
-; CHECK-NEXT:    store <4 x i32> [[TMP3]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 52), align 4
-; CHECK-NEXT:    [[TMP4:%.*]] = load <7 x i32>, ptr getelementptr inbounds nuw (i8, ptr @c, i64 200), align 4
-; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <7 x i32> [[TMP4]], i64 3
+; CHECK-NEXT:    [[ARRAYIDX12_PROMOTED_5_I:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 200), align 4
+; CHECK-NEXT:    [[CONV14_5_I:%.*]] = xor i32 [[ARRAYIDX12_PROMOTED_5_I]], 1
+; CHECK-NEXT:    store i32 [[CONV14_5_I]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 200), align 4
+; CHECK-NEXT:    [[ARRAYIDX12_PROMOTED_5_I_1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 204), align 4
+; CHECK-NEXT:    [[CONV14_5_I_1:%.*]] = xor i32 [[ARRAYIDX12_PROMOTED_5_I_1]], 1
+; CHECK-NEXT:    store i32 [[CONV14_5_I_1]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 204), align 4
+; CHECK-NEXT:    [[TMP1:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 52), align 4
+; CHECK-NEXT:    [[CONV14_1_I_3:%.*]] = xor i32 [[TMP1]], 1
+; CHECK-NEXT:    store i32 [[CONV14_1_I_3]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 52), align 4
+; CHECK-NEXT:    [[ARRAYIDX12_PROMOTED_1_I_4:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 56), align 4
+; CHECK-NEXT:    [[CONV14_1_I_4:%.*]] = xor i32 [[ARRAYIDX12_PROMOTED_1_I_4]], 1
+; CHECK-NEXT:    store i32 [[CONV14_1_I_4]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 56), align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 60), align 4
+; CHECK-NEXT:    [[CONV14_1_I_5:%.*]] = xor i32 [[TMP2]], 1
+; CHECK-NEXT:    store i32 [[CONV14_1_I_5]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 60), align 4
+; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @c, i64 208), align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = extractelement <4 x i32> [[TMP0]], i64 1
 ; CHECK-NEXT:    [[OR_1_5_I_3:%.*]] = or i32 [[TMP1]], [[TMP5]]
 ; CHECK-NEXT:    store i32 [[OR_1_5_I_3]], ptr @j.0, align 4
-; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <7 x i32> [[TMP4]], i64 5
-; CHECK-NEXT:    [[TMP7:%.*]] = xor <7 x i32> [[TMP4]], splat (i32 1)
-; CHECK-NEXT:    store <7 x i32> [[TMP7]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 200), align 4
+; CHECK-NEXT:    [[TMP3:%.*]] = xor <4 x i32> [[TMP0]], splat (i32 1)
+; CHECK-NEXT:    store <4 x i32> [[TMP3]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 208), align 4
+; CHECK-NEXT:    [[TMP6:%.*]] = extractelement <4 x i32> [[TMP0]], i64 3
+; CHECK-NEXT:    [[ARRAYIDX12_PROMOTED_1_I_6:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 64), align 4
+; CHECK-NEXT:    [[CONV14_1_I_6:%.*]] = xor i32 [[ARRAYIDX12_PROMOTED_1_I_6]], 1
+; CHECK-NEXT:    store i32 [[CONV14_1_I_6]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 64), align 4
+; CHECK-NEXT:    [[ARRAYIDX12_PROMOTED_5_I_6:%.*]] = load i32, ptr getelementptr inbounds nuw (i8, ptr @c, i64 224), align 4
+; CHECK-NEXT:    [[CONV14_5_I_6:%.*]] = xor i32 [[ARRAYIDX12_PROMOTED_5_I_6]], 1
+; CHECK-NEXT:    store i32 [[CONV14_5_I_6]], ptr getelementptr inbounds nuw (i8, ptr @c, i64 224), align 4
 ; CHECK-NEXT:    [[TMP8:%.*]] = load <4 x i32>, ptr getelementptr inbounds nuw (i8, ptr @c, i64 252), align 4
 ; CHECK-NEXT:    [[TMP9:%.*]] = extractelement <4 x i32> [[TMP8]], i64 1
 ; CHECK-NEXT:    [[TMP10:%.*]] = or i32 [[TMP9]], [[TMP2]]
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/phi-multi-same-nodes.ll b/llvm/test/Transforms/SLPVectorizer/X86/phi-multi-same-nodes.ll
index 681d91df53b91..0c73c985bf079 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/phi-multi-same-nodes.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/phi-multi-same-nodes.ll
@@ -19,9 +19,6 @@ define void @test() {
 ; CHECK-NEXT:      i32 19, label %[[BB4]]
 ; CHECK-NEXT:    ]
 ; CHECK:       [[BB1]]:
-; CHECK-NEXT:    [[TMP0:%.*]] = ashr <2 x i32> zeroinitializer, <i32 1, i32 0>
-; CHECK-NEXT:    [[TMP1:%.*]] = or <2 x i32> zeroinitializer, <i32 1, i32 0>
-; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[TMP1]], <2 x i32> <i32 0, i32 3>
 ; CHECK-NEXT:    switch i32 0, label %[[BB4]] [
 ; CHECK-NEXT:      i32 -4, label %[[BB4]]
 ; CHECK-NEXT:      i32 -1, label %[[BB4]]
@@ -37,7 +34,7 @@ define void @test() {
 ; CHECK-NEXT:      i32 19, label %[[BB4]]
 ; CHECK-NEXT:    ]
 ; CHECK:       [[BB4]]:
-; CHECK-NEXT:    [[TMP3:%.*]] = phi <2 x i32> [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ [[TMP2]], %[[BB1]] ], [ [[TMP2]], %[[BB1]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = phi <2 x i32> [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB]] ], [ zeroinitializer, %[[BB1]] ], [ zeroinitializer, %[[BB1]] ]
 ; CHECK-NEXT:    ret void
 ;
 bb:

>From b12510424a8b576e3bc2bde27added1f57d70d23 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Mon, 21 Sep 2026 04:39:38 -0700
Subject: [PATCH 2/2] Address comments

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 26 +++++++++----------
 1 file changed, 13 insertions(+), 13 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index ad45c8b917632..5d2bf0cd7df31 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -10331,13 +10331,13 @@ bool BoUpSLP::canBuildSplitNode(ArrayRef<Value *> VL,
       if (I->isBinaryOp())
         Ops1.push_back(I->getOperand(1));
     }
-    TTI::OperandValueInfo Op1Info = getOperandInfo(Ops0);
-    TTI::OperandValueInfo Op2Info =
+    TTI::OperandValueInfo Op0Info = getOperandInfo(Ops0);
+    TTI::OperandValueInfo Op1Info =
         Ops1.empty() ? TTI::OperandValueInfo() : getOperandInfo(Ops1);
     InstructionCost OriginalVecOpsCost =
-        TTI->getArithmeticInstrCost(Opcode0, VecTy, CostKind, Op1Info, Op2Info,
+        TTI->getArithmeticInstrCost(Opcode0, VecTy, CostKind, Op0Info, Op1Info,
                                     {}, LocalState.getMainOp()) +
-        TTI->getArithmeticInstrCost(Opcode1, VecTy, CostKind, Op1Info, Op2Info,
+        TTI->getArithmeticInstrCost(Opcode1, VecTy, CostKind, Op0Info, Op1Info,
                                     {}, LocalState.getAltOp());
     SmallVector<int> OriginalMask(VL.size(), PoisonMaskElem);
     for (unsigned Idx : seq<unsigned>(VL.size())) {
@@ -10349,10 +10349,10 @@ bool BoUpSLP::canBuildSplitNode(ArrayRef<Value *> VL,
         OriginalVecOpsCost + getShuffleCost(*TTI, TTI::SK_PermuteTwoSrc, VecTy,
                                             CostKind, OriginalMask);
     InstructionCost NewVecOpsCost =
-        TTI->getArithmeticInstrCost(Opcode0, Op1VecTy, CostKind, Op1Info,
-                                    Op2Info, {}, LocalState.getMainOp()) +
-        TTI->getArithmeticInstrCost(Opcode1, Op2VecTy, CostKind, Op1Info,
-                                    Op2Info, {}, LocalState.getAltOp());
+        TTI->getArithmeticInstrCost(Opcode0, Op1VecTy, CostKind, Op0Info,
+                                    Op1Info, {}, LocalState.getMainOp()) +
+        TTI->getArithmeticInstrCost(Opcode1, Op2VecTy, CostKind, Op0Info,
+                                    Op1Info, {}, LocalState.getAltOp());
     InstructionCost NewCost =
         NewVecOpsCost + InsertCost +
         (!VectorizableTree.empty() && getRootNode().hasState() &&
@@ -17381,15 +17381,15 @@ BoUpSLP::getEntryCost(const TreeEntry *E, ArrayRef<Value *> VectorizedVals,
         VecCost = ChainCost(E->getOpcode()) + ChainCost(E->getAltOpcode());
       } else if (auto *CI0 = dyn_cast<CmpInst>(VL0)) {
         auto *MaskTy = getWidenedType(Builder.getInt1Ty(), VL.size());
-        TTI::OperandValueInfo Op1Info = getOperandInfo(E->getOperand(0));
-        TTI::OperandValueInfo Op2Info = getOperandInfo(E->getOperand(1));
+        TTI::OperandValueInfo Op0Info = getOperandInfo(E->getOperand(0));
+        TTI::OperandValueInfo Op1Info = getOperandInfo(E->getOperand(1));
         VecCost = TTIRef.getCmpSelInstrCost(E->getOpcode(), VecTy, MaskTy,
                                             CI0->getPredicate(), CostKind,
-                                            Op1Info, Op2Info, VL0);
+                                            Op0Info, Op1Info, VL0);
         VecCost += TTIRef.getCmpSelInstrCost(
             E->getOpcode(), VecTy, MaskTy,
-            cast<CmpInst>(E->getAltOp())->getPredicate(), CostKind, Op1Info,
-            Op2Info, E->getAltOp());
+            cast<CmpInst>(E->getAltOp())->getPredicate(), CostKind, Op0Info,
+            Op1Info, E->getAltOp());
       } else {
         Type *SrcSclTy = E->getMainOp()->getOperand(0)->getType();
         auto *SrcTy = getWidenedType(SrcSclTy, VL.size());



More information about the llvm-commits mailing list