[llvm] [SLP]Emit loop-carried horizontal reductions as loop vector accumulator (PR #221598)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 13 10:22:16 PDT 2026


https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/221598

>From 046466febcb6cbd39aa51bfec96b56b2b9fff134 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sun, 6 Sep 2026 12:16:44 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
 =?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 588 +++++++++++++++++-
 .../SLPVectorizer/AArch64/gather-root.ll      |  15 +-
 .../SLPVectorizer/AArch64/getelementptr.ll    |  42 +-
 .../SLPVectorizer/AArch64/horizontal.ll       |  19 +-
 .../AArch64/loop-accumulator-reduction.ll     |  76 ++-
 .../X86/loop-accumulator-reduction.ll         | 120 ++--
 .../SLPVectorizer/X86/slp-fma-loss.ll         |  20 +-
 7 files changed, 726 insertions(+), 154 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 6d1241fe4408f6..e4ef237f2ecd6b 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -145,6 +145,11 @@ static cl::opt<bool>
 ShouldVectorizeHor("slp-vectorize-hor", cl::init(true), cl::Hidden,
                    cl::desc("Attempt to vectorize horizontal reductions"));
 
+static cl::opt<bool> VectorizeLoopAccRdx(
+    "slp-vectorize-loop-acc-rdx", cl::init(true), cl::Hidden,
+    cl::desc("Emit loop-carried horizontal reductions as a vector "
+             "accumulator in the loop plus a single reduction after it"));
+
 static cl::opt<bool> ShouldStartVectorizeHorAtStore(
     "slp-vectorize-hor-store", cl::init(false), cl::Hidden,
     cl::desc(
@@ -489,6 +494,10 @@ getNumberOfParts(const TargetTransformInfo &TTI, Type *VecTy, Type *ScalarTy,
   return NumParts;
 }
 
+namespace {
+class HorizontalReduction;
+} // namespace
+
 /// Bottom Up SLP Vectorizer.
 class slpvectorizer::BoUpSLP {
   class TreeEntry;
@@ -4343,6 +4352,7 @@ class slpvectorizer::BoUpSLP {
 
   friend struct GraphTraits<BoUpSLP *>;
   friend struct DOTGraphTraits<BoUpSLP *>;
+  friend class ::HorizontalReduction;
 
   /// Contains all scheduling data for a basic block.
   /// It does not schedules instructions, which are not memory read/write
@@ -31052,10 +31062,497 @@ class HorizontalReduction {
     return true;
   }
 
+  /// Loop accumulator emission mode of the reduction: a reduction that
+  /// accumulates a loop phi (the phi is one of the reduced values and the
+  /// root is its backedge value) can be emitted as a vector accumulator in
+  /// the loop plus a single reduction after the loop instead of a horizontal
+  /// reduction on every iteration.
+  class LoopAccumulator {
+    /// The reduction root and the reduction operation kind and ordering.
+    Instruction *Root = nullptr;
+    RecurKind RdxKind = RecurKind::None;
+    ReductionOrdering RK = ReductionOrdering::None;
+    /// The vectorizer state and the analyses the accumulator cost and
+    /// emission are computed with.
+    BoUpSLP &R;
+    const TargetTransformInfo &TTI;
+    LoopInfo &LI;
+    /// The accumulated phi and the identity constant that replaced it in the
+    /// reduced values. The identity is never folded as a leftover: it does
+    /// not change the result.
+    PHINode *Phi = nullptr;
+    Constant *Identity = nullptr;
+    /// The number of calls in the loop, across which the vector accumulator
+    /// is kept live.
+    unsigned NumCalls = 0;
+    /// Exit phis consuming the reduction result outside the loop.
+    SmallSetVector<PHINode *, 2> ExitPhis;
+
+    /// \returns the identity constant of the reduction operation for \p Ty
+    /// under the fast-math flags \p FMF.
+    Constant *getIdentity(Type *Ty, FastMathFlags FMF) const {
+      Intrinsic::ID Id;
+      switch (RdxKind) {
+      case RecurKind::FMax:
+      case RecurKind::FMaxNum:
+        Id = Intrinsic::vector_reduce_fmax;
+        break;
+      case RecurKind::FMin:
+      case RecurKind::FMinNum:
+        Id = Intrinsic::vector_reduce_fmin;
+        break;
+      case RecurKind::FMaximum:
+        Id = Intrinsic::vector_reduce_fmaximum;
+        break;
+      case RecurKind::FMinimum:
+        Id = Intrinsic::vector_reduce_fminimum;
+        break;
+      default:
+        Id = RecurrenceDescriptor::isIntMinMaxRecurrenceKind(RdxKind)
+                 ? getMinMaxReductionIntrinsicID(
+                       getMinMaxReductionIntrinsicOp(RdxKind))
+                 : getReductionForBinop(static_cast<Instruction::BinaryOps>(
+                       RecurrenceDescriptor::getOpcode(RdxKind)));
+        break;
+      }
+      return cast<Constant>(getReductionIdentity(Id, Ty, FMF));
+    }
+
+    /// \returns the extra cost of accumulating the slice lane-wise into the
+    /// vector accumulator instead of reducing it on every iteration: the
+    /// lane-wise operation, the spill and reload on every iteration of the
+    /// values live beyond the register file and keeping the vector
+    /// accumulator live across the calls of the loop.
+    InstructionCost getAccumulationCost(FastMathFlags FMF, VectorType *VectorTy,
+                                        Type *ScalarTy) const {
+      const TTI::TargetCostKind CostKind = R.getCostKind();
+      auto GetOpCost = [&](Type *OpTy) {
+        if (RecurrenceDescriptor::isMinMaxRecurrenceKind(RdxKind)) {
+          IntrinsicCostAttributes ICA(getMinMaxReductionIntrinsicOp(RdxKind),
+                                      OpTy, {OpTy, OpTy}, FMF);
+          return TTI.getIntrinsicInstrCost(ICA, CostKind);
+        }
+        return TTI.getArithmeticInstrCost(
+            RecurrenceDescriptor::getOpcode(RdxKind), OpTy, CostKind);
+      };
+      // The lane-wise operation; also replaces the scalar operation folding
+      // the accumulator phi on top of the horizontal reduction (the scalar
+      // cost of the slice credits it neither, the slice lacks the phi).
+      InstructionCost Cost = GetOpCost(VectorTy) - GetOpCost(ScalarTy);
+      // The vector accumulator and the reduced vector are live at the same
+      // time; the registers they need beyond the register file are spilled
+      // and reloaded on every iteration.
+      constexpr unsigned NumLiveVectors = 2;
+      unsigned Parts = R.getNumberOfParts(VectorTy, ScalarTy);
+      unsigned RC = TTI.getRegisterClassForType(/*Vector=*/true, VectorTy);
+      unsigned NumRegs = TTI.getNumberOfRegisters(RC);
+      InstructionCost SpillReload =
+          TTI.getRegisterClassSpillCost(RC, CostKind) +
+          TTI.getRegisterClassReloadCost(RC, CostKind);
+      if (NumRegs != 0 && NumLiveVectors * Parts > NumRegs)
+        Cost += SpillReload * (NumLiveVectors * Parts - NumRegs);
+      // The vector accumulator replaces the scalar one across the calls of
+      // the loop: without callee-saved vector registers it is spilled and
+      // reloaded around every call, like a scalar FP accumulator.
+      if (NumCalls != 0) {
+        InstructionCost VecLive = std::max(
+            TTI.getCostOfKeepingLiveOverCall(VectorTy), SpillReload * Parts);
+        InstructionCost ScalarLive = TTI.getCostOfKeepingLiveOverCall(ScalarTy);
+        if (ScalarTy->isFloatingPointTy()) {
+          unsigned SRC =
+              TTI.getRegisterClassForType(/*Vector=*/false, ScalarTy);
+          ScalarLive = std::max(
+              ScalarLive, TTI.getRegisterClassSpillCost(SRC, CostKind) +
+                              TTI.getRegisterClassReloadCost(SRC, CostKind));
+        }
+        Cost += (VecLive - ScalarLive) * NumCalls;
+      }
+      return Cost;
+    }
+
+  public:
+    LoopAccumulator(Instruction *Root, RecurKind RdxKind, ReductionOrdering RK,
+                    BoUpSLP &R, const TargetTransformInfo &TTI, LoopInfo &LI)
+        : Root(Root), RdxKind(RdxKind), RK(RK), R(R), TTI(TTI), LI(LI) {}
+
+    /// If the reduction accumulates a loop phi (the phi is one of the reduced
+    /// values and the root is its backedge value), replaces the phi by the
+    /// reduction identity in the reduced values, so that the emitted tree
+    /// does not reference the phi. The identity does not change the result,
+    /// so if the accumulator emission does not apply, the phi is just folded
+    /// back on top of the emitted reduction. All unordered reduction kinds
+    /// have an identity constant.
+    void prepare(FastMathFlags RdxFMF,
+                 SmallVectorImpl<SmallVector<Value *>> &ReducedVals,
+                 SmallDenseMap<Value *, SmallVector<Instruction *>, 16>
+                     &ReducedValsToOps,
+                 bool HasNarrowedLeafShifts) {
+      if (!VectorizeLoopAccRdx || RK != ReductionOrdering::Unordered ||
+          Root->getType()->isIntegerTy(1) || Root->getType()->isVectorTy() ||
+          HasNarrowedLeafShifts || isCmpSelMinMax(Root) ||
+          Root->hasNUsesOrMore(UsesLimit))
+        return;
+      Loop *L = LI.getLoopFor(Root->getParent());
+      if (!L)
+        return;
+      BasicBlock *Latch = L->getLoopLatch();
+      if (!Latch)
+        return;
+      // Only a single accumulated phi, used by the reduction operations
+      // only, is supported.
+      auto FindAccPhi = [&]() -> PHINode * {
+        PHINode *AccPhi = nullptr;
+        for (ArrayRef<Value *> Candidates : ReducedVals)
+          for (Value *RdxVal : Candidates) {
+            auto *P = dyn_cast<PHINode>(RdxVal);
+            if (!P || P->getParent() != L->getHeader() ||
+                P->getNumIncomingValues() > MaxPHINumOperands ||
+                P->getIncomingValueForBlock(Latch) != Root)
+              continue;
+            if (AccPhi || P->hasNUsesOrMore(UsesLimit) ||
+                P->getNumUses() != ReducedValsToOps.at(P).size())
+              return nullptr;
+            AccPhi = P;
+          }
+        return AccPhi;
+      };
+      PHINode *AccPhi = FindAccPhi();
+      if (!AccPhi)
+        return;
+      Constant *IdC = getIdentity(AccPhi->getType(), RdxFMF);
+      if (ReducedValsToOps.contains(IdC))
+        return;
+      // The root may be used only by the accumulator phi and by phis outside
+      // the loop (the exit phis). The initial values are inserted into the
+      // identity vector at the end of their incoming blocks, so they must not
+      // be defined by the terminators; the final reduction is emitted at the
+      // beginning of the exit blocks, which must not be EH pads. Neither may
+      // land in a loop not containing the reduction loop, which would execute
+      // them repeatedly.
+      auto IsExecutedOncePerLoop = [&](BasicBlock *BB) {
+        Loop *BBL = LI.getLoopFor(BB);
+        return !BBL || BBL->contains(L);
+      };
+      if (any_of(zip(AccPhi->incoming_values(), AccPhi->blocks()),
+                 [&](const auto &P) {
+                   auto [IncV, B] = P;
+                   return IncV != Root && (IncV == B->getTerminator() ||
+                                           !IsExecutedOncePerLoop(B));
+                 }))
+        return;
+      auto IsValidExitPhi = [&](User *U) {
+        auto *ExitPhi = dyn_cast<PHINode>(U);
+        return ExitPhi &&
+               ExitPhi->getNumIncomingValues() <= MaxPHINumOperands &&
+               !ExitPhi->getParent()->isEHPad() &&
+               !L->contains(ExitPhi->getParent()) &&
+               IsExecutedOncePerLoop(ExitPhi->getParent());
+      };
+      auto CollectExitPhis = [&] {
+        for (User *U : Root->users()) {
+          if (U == AccPhi)
+            continue;
+          if (!IsValidExitPhi(U))
+            return false;
+          ExitPhis.insert(cast<PHINode>(U));
+        }
+        return true;
+      };
+      if (!CollectExitPhis())
+        return;
+      // The final reduction and the insert of the initial value would be on
+      // the loop-carried chain of an enclosing loop if the reduction result
+      // feeds the initial value of the accumulator through it: keep the
+      // horizontal reduction, whose chain is the scalar accumulator only.
+      auto IsFedByExitPhi = [&](Value *InitV) {
+        auto *InitPhi = dyn_cast<PHINode>(InitV);
+        // Too many incoming values to scan: keep the horizontal reduction.
+        if (InitV == Root || !InitPhi)
+          return false;
+        if (InitPhi->getNumIncomingValues() > MaxPHINumOperands)
+          return true;
+        return ExitPhis.contains(InitPhi) ||
+               any_of(InitPhi->incoming_values(), [&](Value *V) {
+                 return isa<PHINode>(V) && ExitPhis.contains(cast<PHINode>(V));
+               });
+      };
+      if (any_of(AccPhi->incoming_values(), IsFedByExitPhi))
+        return;
+      // Calls clobber the vector registers: the vector accumulator has to be
+      // kept live across them. Intrinsics cheaper than a call are not calls.
+      for (BasicBlock *BB : L->blocks())
+        NumCalls += count_if(*BB, [&](const Instruction &I) {
+          auto *CB = dyn_cast<CallBase>(&I);
+          if (!CB || CB->doesNotReturn())
+            return false;
+          auto *II = dyn_cast<IntrinsicInst>(CB);
+          if (!II)
+            return true;
+          if (II->isAssumeLikeIntrinsic())
+            return false;
+          IntrinsicCostAttributes ICA(II->getIntrinsicID(), *II);
+          return TTI.getIntrinsicInstrCost(ICA, R.getCostKind()) >=
+                 TTI.getCallInstrCost(nullptr, II->getType(), ICA.getArgTypes(),
+                                      R.getCostKind());
+        });
+      for (SmallVector<Value *> &Candidates : ReducedVals)
+        for (Value *&RdxVal : Candidates)
+          if (RdxVal == AccPhi)
+            RdxVal = IdC;
+      ReducedValsToOps.try_emplace(IdC, ReducedValsToOps.lookup(AccPhi));
+      ReducedValsToOps.erase(AccPhi);
+      Phi = AccPhi;
+      Identity = IdC;
+    }
+
+    /// Collects the reduced values that were not vectorized; they are folded
+    /// into the reduction on top of the vectorized part. The accumulator
+    /// identity constant is never folded.
+    void collectLeftovers(
+        const DenseMap<Value *, unsigned> &VectorizedVals,
+        ArrayRef<SmallVector<Value *>> ReducedVals,
+        const SmallDenseMap<Value *, SmallVector<Instruction *>, 16>
+            &ReducedValsToOps,
+        SmallVectorImpl<std::pair<Instruction *, Value *>> &LeftoverReductions)
+        const {
+      SmallPtrSet<Value *, 8> Visited;
+      for (ArrayRef<Value *> Candidates : ReducedVals)
+        for (Value *RdxVal : Candidates) {
+          if (RdxVal == Identity || !Visited.insert(RdxVal).second)
+            continue;
+          unsigned NumOps = VectorizedVals.lookup(RdxVal);
+          for (Instruction *RedOp :
+               ArrayRef(ReducedValsToOps.at(RdxVal)).drop_back(NumOps))
+            LeftoverReductions.emplace_back(RedOp, RdxVal);
+        }
+    }
+
+    /// \returns the cost of the code the vector accumulator emits outside the
+    /// loop: the final reductions in the exit blocks and the insert of the
+    /// initial value, scaled by the enclosing loop nest.
+    InstructionCost getExitCost(FastMathFlags FMF, const Loop *L) const {
+      assert(
+          (RecurrenceDescriptor::isMinMaxRecurrenceKind(RdxKind) ||
+           Instruction::isBinaryOp(RecurrenceDescriptor::getOpcode(RdxKind))) &&
+          "Expected arithmetic or min/max reduction operation");
+      FixedVectorType *VecTy = R.getReductionType();
+      TTI::TargetCostKind CostKind = R.getCostKind();
+      InstructionCost RdxCost =
+          RecurrenceDescriptor::isMinMaxRecurrenceKind(RdxKind)
+              ? TTI.getMinMaxReductionCost(
+                    getMinMaxReductionIntrinsicOp(RdxKind), VecTy, FMF,
+                    CostKind)
+              : TTI.getArithmeticReductionCost(
+                    RecurrenceDescriptor::getOpcode(RdxKind), VecTy, FMF,
+                    CostKind);
+      InstructionCost Cost =
+          RdxCost * ExitPhis.size() +
+          TTI.getVectorInstrCost(Instruction::InsertElement, VecTy, CostKind,
+                                 /*Index=*/0);
+      return Cost * R.getLoopNestScale(L->getParentLoop());
+    }
+
+    /// Checks that the slice can be emitted as the vector accumulator: it is
+    /// the only one (all other reduced values are the identity), not scaled,
+    /// not reduced in-tree, of the root type and fully vectorized. An integer
+    /// accumulator stays in a callee-saved register across the calls of the
+    /// loop, the vector one would be spilled around every call on the
+    /// loop-carried chain.
+    bool isCandidate(const Value *VectorizedTree, bool HasVectorizedSlices,
+                     bool IsSupportedHorRdxIdentityOp,
+                     ArrayRef<SmallVector<Value *>> ReducedVals,
+                     ArrayRef<Value *> Candidates, unsigned I, unsigned Pos,
+                     unsigned ReduxWidth, bool OptReusedScalars,
+                     bool SameScaleFactor) const {
+      if (!Phi || VectorizedTree || HasVectorizedSlices || Pos != 0 ||
+          (OptReusedScalars && SameScaleFactor) || R.isReducedBitcastRoot() ||
+          R.isReducedCmpBitcastRoot() ||
+          (NumCalls != 0 && !Root->getType()->isFloatingPointTy()))
+        return false;
+      if (!all_of(ArrayRef(Candidates).drop_front(ReduxWidth),
+                  [&](Value *RdxVal) { return RdxVal == Identity; }))
+        return false;
+      if (!all_of(enumerate(ReducedVals), [&](const auto &P) {
+            auto [Idx, RV] = P;
+            return Idx == I || (RV.size() == 1 && RV.front() == Identity);
+          }))
+        return false;
+      if (R.getReductionType()->getElementType() != Root->getType())
+        return false;
+      return IsSupportedHorRdxIdentityOp ||
+             all_of(Candidates.slice(Pos, ReduxWidth),
+                    [&](Value *RdxVal) { return R.isVectorized(RdxVal); });
+    }
+
+    /// \returns the cost delta of the vector accumulator form vs the
+    /// horizontal reduction on every iteration. Both remove the same scalar
+    /// operations, so only the vector code is compared: the lane-wise
+    /// operation is executed on every iteration of the loop, the horizontal
+    /// reduction as often as the tree cost model assumes (once, if the
+    /// reduced values are loop-invariant), the final reduction and the
+    /// initial insert once, outside the loop.
+    InstructionCost getCostDelta(InstructionCost HorVecCost,
+                                 InstructionCost RdxOpCost, FastMathFlags FMF,
+                                 VectorType *VectorTy, Type *ScalarTy) const {
+      // The lane-wise accumulation replaces the reduction operation in the
+      // vector code.
+      InstructionCost AccVecCost =
+          HorVecCost - RdxOpCost + getAccumulationCost(FMF, VectorTy, ScalarTy);
+      Loop *L = LI.getLoopFor(Phi->getParent());
+      InstructionCost Delta = AccVecCost * R.getLoopNestScale(L) -
+                              HorVecCost * R.getScaleToLoopIterations(
+                                               R.getRootNode(), nullptr, Root) +
+                              getExitCost(FMF, L);
+      LLVM_DEBUG(dbgs() << "SLP: Loop accumulator cost delta " << Delta
+                        << "\n");
+      return Delta;
+    }
+
+    /// Emits the reduction as a vector accumulator if the accumulator form
+    /// applies; otherwise folds the accumulator phi back on top of the
+    /// reduction like a leftover and returns null.
+    Value *tryEmit(
+        IRBuilderBase &Builder, FastMathFlags RdxFMF, bool Vectorized,
+        const Value *VectorizedTree,
+        SmallVectorImpl<std::pair<Instruction *, Value *>> &LeftoverReductions,
+        const SmallVector<std::tuple<WeakTrackingVH, unsigned, bool, bool>>
+            &VectorValuesAndScales,
+        const SmallPtrSetImpl<Value *> &RequiredExtract,
+        const SmallDenseMap<Value *, SmallVector<Instruction *>, 16>
+            &ReducedValsToOps,
+        const ReductionOpsListType &ReductionOps,
+        function_ref<Value *(Value *, IRBuilderBase &, Type *)> EmitReduction) {
+      if (!shouldEmit(Vectorized, VectorizedTree, LeftoverReductions,
+                      VectorValuesAndScales, RequiredExtract)) {
+        // The accumulator phi is folded back on top of the reduction like a
+        // leftover: the identity replaced it in the reduced values and does
+        // not change the result.
+        if (Phi)
+          LeftoverReductions.emplace_back(ReducedValsToOps.at(Identity).front(),
+                                          Phi);
+        return nullptr;
+      }
+      return emit(Builder, RdxFMF, std::get<0>(VectorValuesAndScales.front()),
+                  ReductionOps, EmitReduction);
+    }
+
+  private:
+    /// Checks that the accumulator form applies: the slice was vectorized as
+    /// the accumulator and no other vectorized value or leftover reduced
+    /// values remain to fold; the slice is of the root type, not scaled, not
+    /// reduced in-tree and without extracts.
+    bool shouldEmit(
+        bool Vectorized, const Value *VectorizedTree,
+        ArrayRef<std::pair<Instruction *, Value *>> LeftoverReductions,
+        const SmallVector<std::tuple<WeakTrackingVH, unsigned, bool, bool>>
+            &VectorValuesAndScales,
+        const SmallPtrSetImpl<Value *> &RequiredExtract) const {
+      if (!Vectorized || VectorizedTree || !LeftoverReductions.empty() ||
+          VectorValuesAndScales.size() != 1 || !RequiredExtract.empty())
+        return false;
+      const auto &[Vec, Scale, IsSigned, ReducedInTree] =
+          VectorValuesAndScales.front();
+      return Scale == 1 && !ReducedInTree &&
+             cast<VectorType>(Vec->getType())->getElementType() ==
+                 Root->getType();
+    }
+
+    /// Emits the reduction as a vector accumulator: the vectorized reduced
+    /// values are accumulated lane-wise into a vector phi in the loop and
+    /// reduced once in the exit blocks.
+    Value *emit(
+        IRBuilderBase &Builder, FastMathFlags RdxFMF, Value *Vec,
+        const ReductionOpsListType &ReductionOps,
+        function_ref<Value *(Value *, IRBuilderBase &, Type *)> EmitReduction) {
+      auto *VecTy = cast<FixedVectorType>(Vec->getType());
+      // The initial values are placed in lane 0 of the identity splat at the
+      // end of their incoming blocks; multiple edges from a block carry the
+      // same value and share the insert.
+      Constant *IdVec =
+          ConstantVector::getSplat(VecTy->getElementCount(), Identity);
+      auto *VAcc = PHINode::Create(VecTy, Phi->getNumIncomingValues(),
+                                   "slprdx.acc", Phi->getIterator());
+      SmallDenseMap<BasicBlock *, Value *> InitVecs;
+      for (auto [IncV, B] : zip(Phi->incoming_values(), Phi->blocks())) {
+        if (IncV == Root)
+          continue;
+        Value *&InitVec = InitVecs[B];
+        if (!InitVec) {
+          IRBuilder<> EB(B->getTerminator());
+          InitVec = EB.CreateInsertElement(IdVec, IncV, EB.getInt32(0),
+                                           "slprdx.init");
+        }
+        VAcc->addIncoming(InitVec, B);
+      }
+      Builder.SetInsertPoint(Root);
+      Value *VAdd =
+          createOp(Builder, RdxKind, VAcc, Vec, "slprdx.acc", ReductionOps);
+      // The accumulation is on the loop-carried chain: do not let it be
+      // contracted into an FMA with the reduced multiplication, which would
+      // lengthen the chain.
+      if (auto *FPOp = dyn_cast<FPMathOperator>(VAdd)) {
+        FastMathFlags FMF = FPOp->getFastMathFlags();
+        FMF.setAllowContract(false);
+        cast<Instruction>(VAdd)->copyFastMathFlags(FMF);
+      }
+      // The accumulator is the backedge value.
+      for (auto [IncV, B] : zip(Phi->incoming_values(), Phi->blocks()))
+        if (IncV == Root)
+          VAcc->addIncoming(VAdd, B);
+      LLVM_DEBUG(dbgs() << "SLP: Emitting loop accumulator reduction for "
+                           "reduction with root "
+                        << *Root << "\n");
+      // The root is tracked by a weak handle: break its uses on the phi side.
+      Value *Poison = PoisonValue::get(Root->getType());
+      for (PHINode *ExitPhi : ExitPhis) {
+        // The accumulator replaces the root on the edges carrying it and is
+        // reduced once in the exit block. The values bypassing the loop never
+        // went through the reduction operations and must stay exact: the
+        // scalar phi keeps them and they are selected past the reduction.
+        unsigned NumIncoming = ExitPhi->getNumIncomingValues();
+        auto *VExit = PHINode::Create(VecTy, NumIncoming, "slprdx.exit",
+                                      ExitPhi->getIterator());
+        PHINode *FromLoop = nullptr;
+        if (any_of(ExitPhi->incoming_values(),
+                   [this](Value *V) { return V != Root; }))
+          FromLoop = PHINode::Create(Builder.getInt1Ty(), NumIncoming,
+                                     "slprdx.fromloop", ExitPhi->getIterator());
+        Value *VecPoison = PoisonValue::get(VecTy);
+        for (auto [V, B] : zip(ExitPhi->incoming_values(), ExitPhi->blocks())) {
+          bool IsRoot = V == Root;
+          VExit->addIncoming(IsRoot ? VAdd : VecPoison, B);
+          if (FromLoop)
+            FromLoop->addIncoming(Builder.getInt1(IsRoot), B);
+        }
+        ExitPhi->replaceUsesOfWith(Root, Poison);
+        IRBuilder<> XB(&*ExitPhi->getParent()->getFirstInsertionPt());
+        XB.SetCurrentDebugLocation(ExitPhi->getDebugLoc());
+        XB.setFastMathFlags(RdxFMF);
+        Value *Res = EmitReduction(VExit, XB, ExitPhi->getType());
+        if (!FromLoop) {
+          ExitPhi->replaceAllUsesWith(Res);
+          // The erasure is deferred: the block iteration is still in progress.
+          R.eraseInstruction(ExitPhi);
+          continue;
+        }
+        Value *Sel = XB.CreateSelectWithUnknownProfile(
+            FromLoop, Res, ExitPhi, DEBUG_TYPE, "slprdx.sel");
+        ExitPhi->replaceUsesWithIf(
+            Sel, [Sel](Use &U) { return U.getUser() != Sel; });
+      }
+      Phi->replaceUsesOfWith(Root, Poison);
+      R.eraseInstruction(Phi);
+      // Do not let the caller re-analyze the emitted vector operation as a new
+      // reduction seed.
+      R.analyzedReductionRoot(cast<Instruction>(VAdd));
+      return VAdd;
+    }
+  };
+
   /// Attempt to vectorize the tree found by matchAssociativeReduction.
   Value *tryToReduce(BoUpSLP &V, const DataLayout &DL, TargetTransformInfo *TTI,
                      const TargetLibraryInfo &TLI, AssumptionCache *AC,
-                     DominatorTree &DT) {
+                     DominatorTree &DT, LoopInfo &LI) {
     constexpr unsigned RegMaxNumber = 4;
     const unsigned RedValsMaxNumber =
         (RK == ReductionOrdering::Ordered &&
@@ -31171,6 +31668,15 @@ class HorizontalReduction {
       IgnoreList.clear();
     bool IsCmpSelMinMax = isCmpSelMinMax(cast<Instruction>(ReductionRoot));
 
+    // A reduction that accumulates a loop phi can be emitted as a vector
+    // accumulator in the loop plus a single reduction after it.
+    LoopAccumulator LoopAcc(cast<Instruction>(ReductionRoot), RdxKind, RK, V,
+                            *TTI, LI);
+    LoopAcc.prepare(RdxFMF, ReducedVals, ReducedValsToOps,
+                    !NarrowedLeafShifts.empty());
+    // Set when the slice costed as the vector accumulator was vectorized.
+    bool LoopAccVectorized = false;
+
     // Need to track reduced vals, they may be changed during vectorization of
     // subvectors.
     for (ArrayRef<Value *> Candidates : ReducedVals)
@@ -31542,6 +32048,10 @@ class HorizontalReduction {
         V.transformNodes();
         V.computeMinimumValueSizes();
         InstructionCost TreeCost = V.calculateTreeCostAndTrimNonProfitable(VL);
+        const bool LoopAccCandidate = LoopAcc.isCandidate(
+            VectorizedTree, !VectorValuesAndScales.empty(),
+            IsSupportedHorRdxIdentityOp, ReducedVals, Candidates, I, Pos,
+            ReduxWidth, OptReusedScalars, SameScaleFactor);
 
         SmallPtrSet<Value *, 4> VLScalars(llvm::from_range, VL);
         // Gather externally used values.
@@ -31575,14 +32085,25 @@ class HorizontalReduction {
         V.buildExternalUses(LocalExternallyUsedValues);
 
         // Estimate cost.
-        InstructionCost ReductionCost;
+        InstructionCost ReductionCost, HorVecCost = 0, RdxOpCost = 0;
         if (RK == ReductionOrdering::Ordered || V.isReducedBitcastRoot() ||
             V.isReducedCmpBitcastRoot())
           ReductionCost = 0;
         else
           ReductionCost =
               getReductionCost(TTI, VL, SameValuesCounter, IsCmpSelMinMax,
-                               RdxFMF, V, DT, DL, TLI);
+                               RdxFMF, V, DT, DL, TLI, HorVecCost, RdxOpCost);
+        // The vector accumulator form is used if it is not more expensive than
+        // the horizontal reduction on every iteration.
+        InstructionCost LoopAccDelta = 0;
+        if (LoopAccCandidate && ReductionCost.isValid())
+          LoopAccDelta =
+              LoopAcc.getCostDelta(HorVecCost, RdxOpCost, RdxFMF,
+                                   V.getReductionType(), VL.front()->getType());
+        // On a tie the accumulator wins: it also replaces the loop-carried
+        // scalar dependency through the horizontal reduction by a lane-wise
+        // one.
+        const bool UseLoopAccForm = LoopAccCandidate && LoopAccDelta <= 0;
         // If the root is a select (min/max idiom), the insert point is the
         // compare condition of that select.
         Instruction *RdxRootInst = cast<Instruction>(ReductionRoot);
@@ -31591,6 +32112,8 @@ class HorizontalReduction {
           InsertPt = GetCmpForMinMaxReduction(RdxRootInst);
         InstructionCost Cost =
             V.getTreeCost(TreeCost, VL, ReductionCost, InsertPt);
+        if (UseLoopAccForm)
+          Cost += LoopAccDelta;
         LLVM_DEBUG(dbgs() << "SLP: Found cost = " << Cost
                           << " for reduction\n");
         if (!Cost.isValid())
@@ -31729,6 +32252,7 @@ class HorizontalReduction {
                 ? NarrowedLeafShifts.empty() && V.isSignedMinBitwidthRootNode()
                 : true,
             V.isReducedBitcastRoot() || V.isReducedCmpBitcastRoot());
+        LoopAccVectorized = UseLoopAccForm;
 
         // Count vectorized reduced values to exclude them from final reduction.
         for (const auto [Idx, RdxVal] : enumerate(VL)) {
@@ -31747,6 +32271,10 @@ class HorizontalReduction {
         if (ReduxWidth > 1)
           ReduxWidth = GetVectorFactor(NumReducedVals - Pos);
         AnyVectorized = true;
+        // All remaining reduced values are the identity: nothing left to
+        // vectorize for the accumulator form.
+        if (UseLoopAccForm)
+          break;
       }
       if (OptReusedScalars && !AnyVectorized) {
         for (const std::pair<Value *, unsigned> &P : SameValuesCounter) {
@@ -31763,7 +32291,23 @@ class HorizontalReduction {
     if (RK == ReductionOrdering::Ordered)
       return VectorizedTree;
 
-    if (!VectorValuesAndScales.empty())
+    SmallVector<std::pair<Instruction *, Value *>> LeftoverReductions;
+    LoopAcc.collectLeftovers(VectorizedVals, ReducedVals, ReducedValsToOps,
+                             LeftoverReductions);
+
+    // The accumulator form requires a single vectorized slice and no
+    // leftover reduced values; the latter would have to be folded into the
+    // scalar accumulator on every iteration. Otherwise the accumulator phi is
+    // folded back on top of the reduction like a leftover.
+    Value *AccV = LoopAcc.tryEmit(
+        Builder, RdxFMF, LoopAccVectorized, VectorizedTree, LeftoverReductions,
+        VectorValuesAndScales, RequiredExtract, ReducedValsToOps, ReductionOps,
+        [this, TTI](Value *Vec, IRBuilderBase &B, Type *Ty) {
+          return emitReduction(Vec, B, TTI, Ty);
+        });
+    if (AccV)
+      VectorizedTree = AccV;
+    else if (!VectorValuesAndScales.empty())
       VectorizedTree = GetNewVectorizedTree(
           VectorizedTree,
           emitReduction(Builder, *TTI, ReductionRoot->getType()));
@@ -31877,17 +32421,7 @@ class HorizontalReduction {
     SmallVector<std::pair<Instruction *, Value *>> ExtraReductions;
     ExtraReductions.emplace_back(cast<Instruction>(ReductionRoot),
                                  VectorizedTree);
-    SmallPtrSet<Value *, 8> Visited;
-    for (ArrayRef<Value *> Candidates : ReducedVals) {
-      for (Value *RdxVal : Candidates) {
-        if (!Visited.insert(RdxVal).second)
-          continue;
-        unsigned NumOps = VectorizedVals.lookup(RdxVal);
-        for (Instruction *RedOp :
-             ArrayRef(ReducedValsToOps.at(RdxVal)).drop_back(NumOps))
-          ExtraReductions.emplace_back(RedOp, RdxVal);
-      }
-    }
+    ExtraReductions.append(LeftoverReductions);
     // Iterate through all not-vectorized reduction values/extra arguments.
     bool InitStep = true;
     while (ExtraReductions.size() > 1) {
@@ -31898,7 +32432,9 @@ class HorizontalReduction {
     }
     VectorizedTree = ExtraReductions.front().second;
 
-    ReductionRoot->replaceAllUsesWith(VectorizedTree);
+    // The accumulator emission already replaced all uses of the root.
+    if (!AccV)
+      ReductionRoot->replaceAllUsesWith(VectorizedTree);
 
     // The original scalar reduction is expected to have no remaining
     // uses outside the reduction tree itself.  Assert that we got this
@@ -32053,9 +32589,11 @@ class HorizontalReduction {
           V.calculateTreeCostAndTrimNonProfitable(VL, RdxRootInst);
       V.buildExternalUses(LocalExternallyUsedValues);
 
+      InstructionCost VectorCost, RdxOpCost;
       InstructionCost ReductionCost =
           getReductionCost(TTI, VL, EmptySameValuesCounter,
-                           /*IsCmpSelMinMax=*/false, RdxFMF, V, DT, DL, TLI);
+                           /*IsCmpSelMinMax=*/false, RdxFMF, V, DT, DL, TLI,
+                           VectorCost, RdxOpCost);
       InstructionCost Cost =
           V.getTreeCost(TreeCost, VL, ReductionCost, RdxRootInst);
       LLVM_DEBUG(dbgs() << "SLP: Found cost = " << Cost
@@ -32271,16 +32809,21 @@ class HorizontalReduction {
   }
 
   /// Calculate the cost of a reduction.
+  /// \p VectorCost receives the cost of the emitted vector code alone and
+  /// \p RdxOpCost the cost of the reduction operation in it.
   InstructionCost getReductionCost(
       TargetTransformInfo *TTI, ArrayRef<Value *> ReducedVals,
       const SmallMapVector<Value *, unsigned, 16> SameValuesCounter,
       bool IsCmpSelMinMax, FastMathFlags FMF, const BoUpSLP &R,
-      DominatorTree &DT, const DataLayout &DL, const TargetLibraryInfo &TLI) {
+      DominatorTree &DT, const DataLayout &DL, const TargetLibraryInfo &TLI,
+      InstructionCost &VectorCost, InstructionCost &RdxOpCost) {
     const TTI::TargetCostKind CostKind = R.getCostKind();
     Type *ScalarTy = ReducedVals.front()->getType();
     unsigned ReduxWidth = ReducedVals.size();
     FixedVectorType *VectorTy = R.getReductionType();
-    InstructionCost VectorCost = 0, ScalarCost;
+    InstructionCost ScalarCost;
+    VectorCost = 0;
+    RdxOpCost = 0;
     // If all of the reduced values are constant, the vector cost is 0, since
     // the reduction value can be calculated at the compile time.
     bool AllConsts = allConstant(ReducedVals);
@@ -32427,6 +32970,7 @@ class HorizontalReduction {
                   cast<VectorType>(getWidenedType(RType, ReduxWidth)), FMF,
                   CostKind);
             }
+            RdxOpCost = VectorCost;
           }
         } else {
           Type *RedTy = VectorTy->getElementType();
@@ -32504,6 +33048,7 @@ class HorizontalReduction {
       if (!AllConsts) {
         if (DoesRequireReductionOp) {
           VectorCost = TTI->getMinMaxReductionCost(Id, VectorTy, FMF, CostKind);
+          RdxOpCost = VectorCost;
         } else {
           // Check if the previous reduction already exists and account it as
           // series of operations + single reduction.
@@ -32521,6 +33066,7 @@ class HorizontalReduction {
             VectorCost += TTI->getCastInstrCost(
                 Opcode, VectorTy, RVecTy, TTI::CastContextHint::None, CostKind);
           }
+          RdxOpCost = VectorCost;
         }
       }
       ScalarCost = EvaluateScalarCost([&](Instruction *RdxOp) {
@@ -33170,7 +33716,7 @@ bool SLPVectorizerPass::vectorizeHorReduction(
     HorizontalReduction HorRdx;
     Value *Res = nullptr;
     if (HorRdx.matchAssociativeReduction(R, Inst, *SE, *DT, *DL, *TTI, *TLI))
-      if (Value *Red = HorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT)) {
+      if (Value *Red = HorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT, *LI)) {
         if (Red != Inst)
           return Red;
         Res = Red;
@@ -33339,7 +33885,7 @@ bool SLPVectorizerPass::tryToVectorize(
     if (RedCost >= ScalarCost)
       return false;
 
-    return HorRdx.tryToReduce(R, *DL, &TTI, *TLI, AC, *DT) != nullptr;
+    return HorRdx.tryToReduce(R, *DL, &TTI, *TLI, AC, *DT, *LI) != nullptr;
   };
   if (Candidates.size() == 1)
     return TryToReduce(I, {Op0, Op1}) ||
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/gather-root.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/gather-root.ll
index 7ae336e2ccee98..681571a54f052c 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/gather-root.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/gather-root.ll
@@ -15,10 +15,9 @@ define void @PR28330(i32 %n) {
 ; DEFAULT-NEXT:    [[TMP1:%.*]] = icmp eq <8 x i8> [[TMP0]], zeroinitializer
 ; DEFAULT-NEXT:    br label [[FOR_BODY:%.*]]
 ; DEFAULT:       for.body:
-; DEFAULT-NEXT:    [[P17:%.*]] = phi i32 [ [[OP_RDX:%.*]], [[FOR_BODY]] ], [ 0, [[ENTRY:%.*]] ]
+; DEFAULT-NEXT:    [[SLPRDX_ACC:%.*]] = phi <8 x i32> [ zeroinitializer, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_BODY]] ]
 ; DEFAULT-NEXT:    [[TMP2:%.*]] = select <8 x i1> [[TMP1]], <8 x i32> splat (i32 -720), <8 x i32> splat (i32 -80)
-; DEFAULT-NEXT:    [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP2]])
-; DEFAULT-NEXT:    [[OP_RDX]] = add i32 [[TMP3]], [[P17]]
+; DEFAULT-NEXT:    [[SLPRDX_ACC1]] = add <8 x i32> [[SLPRDX_ACC]], [[TMP2]]
 ; DEFAULT-NEXT:    br label [[FOR_BODY]]
 ;
 ; GATHER-LABEL: @PR28330(
@@ -27,10 +26,9 @@ define void @PR28330(i32 %n) {
 ; GATHER-NEXT:    [[TMP1:%.*]] = icmp eq <8 x i8> [[TMP0]], zeroinitializer
 ; GATHER-NEXT:    br label [[FOR_BODY:%.*]]
 ; GATHER:       for.body:
-; GATHER-NEXT:    [[P17:%.*]] = phi i32 [ [[OP_RDX:%.*]], [[FOR_BODY]] ], [ 0, [[ENTRY:%.*]] ]
+; GATHER-NEXT:    [[SLPRDX_ACC:%.*]] = phi <8 x i32> [ zeroinitializer, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_BODY]] ]
 ; GATHER-NEXT:    [[TMP2:%.*]] = select <8 x i1> [[TMP1]], <8 x i32> splat (i32 -720), <8 x i32> splat (i32 -80)
-; GATHER-NEXT:    [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP2]])
-; GATHER-NEXT:    [[OP_RDX]] = add i32 [[TMP3]], [[P17]]
+; GATHER-NEXT:    [[SLPRDX_ACC1]] = add <8 x i32> [[SLPRDX_ACC]], [[TMP2]]
 ; GATHER-NEXT:    br label [[FOR_BODY]]
 ;
 ; MAX-COST-LABEL: @PR28330(
@@ -39,10 +37,9 @@ define void @PR28330(i32 %n) {
 ; MAX-COST-NEXT:    [[TMP1:%.*]] = icmp eq <8 x i8> [[TMP0]], zeroinitializer
 ; MAX-COST-NEXT:    br label [[FOR_BODY:%.*]]
 ; MAX-COST:       for.body:
-; MAX-COST-NEXT:    [[P17:%.*]] = phi i32 [ [[OP_RDX:%.*]], [[FOR_BODY]] ], [ 0, [[ENTRY:%.*]] ]
+; MAX-COST-NEXT:    [[SLPRDX_ACC:%.*]] = phi <8 x i32> [ zeroinitializer, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_BODY]] ]
 ; MAX-COST-NEXT:    [[TMP2:%.*]] = select <8 x i1> [[TMP1]], <8 x i32> splat (i32 -720), <8 x i32> splat (i32 -80)
-; MAX-COST-NEXT:    [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP2]])
-; MAX-COST-NEXT:    [[OP_RDX]] = add i32 [[TMP3]], [[P17]]
+; MAX-COST-NEXT:    [[SLPRDX_ACC1]] = add <8 x i32> [[SLPRDX_ACC]], [[TMP2]]
 ; MAX-COST-NEXT:    br label [[FOR_BODY]]
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
index 6df79461fb9144..d7f5ad7861034d 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
@@ -53,12 +53,16 @@ define i32 @getelementptr_4x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y
 ; CHECK:       for.cond.cleanup.loopexit:
 ; CHECK-NEXT:    br label [[FOR_COND_CLEANUP]]
 ; CHECK:       for.cond.cleanup:
-; CHECK-NEXT:    [[SUM_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[ADD16:%.*]], [[FOR_COND_CLEANUP_LOOPEXIT:%.*]] ]
-; CHECK-NEXT:    ret i32 [[SUM_0_LCSSA]]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x i32> [ poison, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_COND_CLEANUP_LOOPEXIT:%.*]] ]
+; CHECK-NEXT:    [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ false, [[ENTRY]] ], [ true, [[FOR_COND_CLEANUP_LOOPEXIT]] ]
+; CHECK-NEXT:    [[SUM_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ poison, [[FOR_COND_CLEANUP_LOOPEXIT]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[SLPRDX_EXIT]])
+; CHECK-NEXT:    [[SLPRDX_SEL:%.*]] = select i1 [[SLPRDX_FROMLOOP]], i32 [[TMP6]], i32 [[SUM_0_LCSSA]]
+; CHECK-NEXT:    ret i32 [[SLPRDX_SEL]]
 ; CHECK:       for.body:
-; CHECK-NEXT:    [[TMP15:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT:    [[SUM_032:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[ADD16]], [[FOR_BODY]] ]
-; CHECK-NEXT:    [[T4:%.*]] = shl nsw i32 [[TMP15]], 1
+; CHECK-NEXT:    [[SUM_32:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[OP_RDX:%.*]], [[FOR_BODY]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x i32> [ zeroinitializer, [[FOR_BODY_PREHEADER]] ], [ [[SLPRDX_ACC1]], [[FOR_BODY]] ]
+; CHECK-NEXT:    [[T4:%.*]] = shl nsw i32 [[SUM_32]], 1
 ; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x i32> poison, i32 [[T4]], i64 0
 ; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP3:%.*]] = add nsw <2 x i32> [[TMP2]], [[TMP0]]
@@ -79,10 +83,9 @@ define i32 @getelementptr_4x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y
 ; CHECK-NEXT:    [[TMP18:%.*]] = insertelement <4 x i32> [[TMP17]], i32 [[T8]], i64 1
 ; CHECK-NEXT:    [[TMP19:%.*]] = insertelement <4 x i32> [[TMP18]], i32 [[T10]], i64 2
 ; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <4 x i32> [[TMP19]], i32 [[T12]], i64 3
-; CHECK-NEXT:    [[TMP10:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP9]])
-; CHECK-NEXT:    [[ADD16]] = add i32 [[TMP10]], [[SUM_032]]
-; CHECK-NEXT:    [[INDVARS_IV_NEXT]] = add nuw nsw i32 [[TMP15]], 1
-; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp eq i32 [[INDVARS_IV_NEXT]], [[N]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = add <4 x i32> [[SLPRDX_ACC]], [[TMP9]]
+; CHECK-NEXT:    [[OP_RDX]] = add nuw nsw i32 [[SUM_32]], 1
+; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp eq i32 [[OP_RDX]], [[N]]
 ; CHECK-NEXT:    br i1 [[EXITCOND]], label [[FOR_COND_CLEANUP_LOOPEXIT]], label [[FOR_BODY]]
 ;
 entry:
@@ -146,12 +149,16 @@ define i32 @getelementptr_2x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y
 ; CHECK:       for.cond.cleanup.loopexit:
 ; CHECK-NEXT:    br label [[FOR_COND_CLEANUP]]
 ; CHECK:       for.cond.cleanup:
-; CHECK-NEXT:    [[SUM_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[OP_RDX:%.*]], [[FOR_COND_CLEANUP_LOOPEXIT:%.*]] ]
-; CHECK-NEXT:    ret i32 [[SUM_0_LCSSA]]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x i32> [ poison, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_COND_CLEANUP_LOOPEXIT:%.*]] ]
+; CHECK-NEXT:    [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ false, [[ENTRY]] ], [ true, [[FOR_COND_CLEANUP_LOOPEXIT]] ]
+; CHECK-NEXT:    [[SUM_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ poison, [[FOR_COND_CLEANUP_LOOPEXIT]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[SLPRDX_EXIT]])
+; CHECK-NEXT:    [[SLPRDX_SEL:%.*]] = select i1 [[SLPRDX_FROMLOOP]], i32 [[TMP4]], i32 [[SUM_0_LCSSA]]
+; CHECK-NEXT:    ret i32 [[SLPRDX_SEL]]
 ; CHECK:       for.body:
-; CHECK-NEXT:    [[TMP12:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT:    [[SUM_032:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[OP_RDX]], [[FOR_BODY]] ]
-; CHECK-NEXT:    [[T4:%.*]] = shl nsw i32 [[TMP12]], 1
+; CHECK-NEXT:    [[SUM_32:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[OP_RDX1:%.*]], [[FOR_BODY]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x i32> [ zeroinitializer, [[FOR_BODY_PREHEADER]] ], [ [[SLPRDX_ACC1]], [[FOR_BODY]] ]
+; CHECK-NEXT:    [[T4:%.*]] = shl nsw i32 [[SUM_32]], 1
 ; CHECK-NEXT:    [[TMP1:%.*]] = insertelement <2 x i32> poison, i32 [[T4]], i64 0
 ; CHECK-NEXT:    [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> zeroinitializer
 ; CHECK-NEXT:    [[TMP3:%.*]] = add nsw <2 x i32> [[TMP2]], [[TMP0]]
@@ -168,10 +175,9 @@ define i32 @getelementptr_2x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y
 ; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <4 x i32> [[TMP7]], i32 [[T12]], i64 3
 ; CHECK-NEXT:    [[TMP13:%.*]] = shufflevector <2 x i32> [[TMP5]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
 ; CHECK-NEXT:    [[TMP14:%.*]] = shufflevector <4 x i32> [[TMP8]], <4 x i32> [[TMP13]], <4 x i32> <i32 4, i32 5, i32 2, i32 3>
-; CHECK-NEXT:    [[TMP11:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP14]])
-; CHECK-NEXT:    [[OP_RDX]] = add i32 [[TMP11]], [[SUM_032]]
-; CHECK-NEXT:    [[INDVARS_IV_NEXT]] = add nuw nsw i32 [[TMP12]], 1
-; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp eq i32 [[INDVARS_IV_NEXT]], [[N]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = add <4 x i32> [[SLPRDX_ACC]], [[TMP14]]
+; CHECK-NEXT:    [[OP_RDX1]] = add nuw nsw i32 [[SUM_32]], 1
+; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp eq i32 [[OP_RDX1]], [[N]]
 ; CHECK-NEXT:    br i1 [[EXITCOND]], label [[FOR_COND_CLEANUP_LOOPEXIT]], label [[FOR_BODY]]
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/horizontal.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/horizontal.ll
index 80168812f5fe6b..d6ec397aa6c2ce 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/horizontal.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/horizontal.ll
@@ -28,8 +28,8 @@ define i32 @test_select(ptr noalias nocapture readonly %blk1, ptr noalias nocapt
 ; CHECK-NEXT:    [[IDX_EXT:%.*]] = sext i32 [[LX:%.*]] to i64
 ; CHECK-NEXT:    br label [[FOR_BODY:%.*]]
 ; CHECK:       for.body:
-; CHECK-NEXT:    [[S_026:%.*]] = phi i32 [ 0, [[FOR_BODY_LR_PH]] ], [ [[OP_RDX:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT:    [[J_025:%.*]] = phi i32 [ 0, [[FOR_BODY_LR_PH]] ], [ [[INC:%.*]], [[FOR_BODY]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x i32> [ zeroinitializer, [[FOR_BODY_LR_PH]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_BODY]] ]
+; CHECK-NEXT:    [[J_25:%.*]] = phi i32 [ 0, [[FOR_BODY_LR_PH]] ], [ [[INC1:%.*]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[P2_024:%.*]] = phi ptr [ [[BLK2:%.*]], [[FOR_BODY_LR_PH]] ], [ [[ADD_PTR29:%.*]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[P1_023:%.*]] = phi ptr [ [[BLK1:%.*]], [[FOR_BODY_LR_PH]] ], [ [[ADD_PTR:%.*]], [[FOR_BODY]] ]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i32>, ptr [[P1_023]], align 4
@@ -38,18 +38,21 @@ define i32 @test_select(ptr noalias nocapture readonly %blk1, ptr noalias nocapt
 ; CHECK-NEXT:    [[TMP3:%.*]] = icmp slt <4 x i32> [[TMP2]], zeroinitializer
 ; CHECK-NEXT:    [[TMP4:%.*]] = sub nsw <4 x i32> zeroinitializer, [[TMP2]]
 ; CHECK-NEXT:    [[TMP5:%.*]] = select <4 x i1> [[TMP3]], <4 x i32> [[TMP4]], <4 x i32> [[TMP2]]
-; CHECK-NEXT:    [[TMP6:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP5]])
-; CHECK-NEXT:    [[OP_RDX]] = add i32 [[TMP6]], [[S_026]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = add <4 x i32> [[SLPRDX_ACC]], [[TMP5]]
 ; CHECK-NEXT:    [[ADD_PTR]] = getelementptr inbounds i32, ptr [[P1_023]], i64 [[IDX_EXT]]
 ; CHECK-NEXT:    [[ADD_PTR29]] = getelementptr inbounds i32, ptr [[P2_024]], i64 [[IDX_EXT]]
-; CHECK-NEXT:    [[INC]] = add nuw nsw i32 [[J_025]], 1
-; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp eq i32 [[INC]], [[H]]
+; CHECK-NEXT:    [[INC1]] = add nuw nsw i32 [[J_25]], 1
+; CHECK-NEXT:    [[EXITCOND:%.*]] = icmp eq i32 [[INC1]], [[H]]
 ; CHECK-NEXT:    br i1 [[EXITCOND]], label [[FOR_END_LOOPEXIT:%.*]], label [[FOR_BODY]]
 ; CHECK:       for.end.loopexit:
 ; CHECK-NEXT:    br label [[FOR_END]]
 ; CHECK:       for.end:
-; CHECK-NEXT:    [[S_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[OP_RDX]], [[FOR_END_LOOPEXIT]] ]
-; CHECK-NEXT:    ret i32 [[S_0_LCSSA]]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x i32> [ poison, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1]], [[FOR_END_LOOPEXIT]] ]
+; CHECK-NEXT:    [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ false, [[ENTRY]] ], [ true, [[FOR_END_LOOPEXIT]] ]
+; CHECK-NEXT:    [[S_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ poison, [[FOR_END_LOOPEXIT]] ]
+; CHECK-NEXT:    [[TMP6:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[SLPRDX_EXIT]])
+; CHECK-NEXT:    [[SLPRDX_SEL:%.*]] = select i1 [[SLPRDX_FROMLOOP]], i32 [[TMP6]], i32 [[S_0_LCSSA]]
+; CHECK-NEXT:    ret i32 [[SLPRDX_SEL]]
 ;
 entry:
   %cmp.22 = icmp sgt i32 %h, 0
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/loop-accumulator-reduction.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/loop-accumulator-reduction.ll
index 256d16254ea292..3ab8ef2410e2db 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/loop-accumulator-reduction.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/loop-accumulator-reduction.ll
@@ -12,7 +12,7 @@ define double @loop_acc_fadd(ptr %p, i64 %n) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[TMP24:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[IV3:%.*]] = mul nuw i64 [[IV]], 3
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV3]]
 ; CHECK-NEXT:    [[P1:%.*]] = getelementptr double, ptr [[P0]], i64 1
@@ -35,30 +35,30 @@ define double @loop_acc_fadd(ptr %p, i64 %n) {
 ; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x double> [[TMP2]], double [[C3]], i64 1
 ; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x double> [[TMP3]], double [[C6]], i64 2
 ; CHECK-NEXT:    [[TMP5:%.*]] = fmul fast <4 x double> [[TMP1]], [[TMP4]]
-; CHECK-NEXT:    [[TMP11:%.*]] = insertelement <4 x double> poison, double [[L1]], i64 0
-; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <4 x double> [[TMP11]], double [[ACC]], i64 1
+; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <4 x double> <double poison, double 0.000000e+00, double poison, double poison>, double [[L1]], i64 0
 ; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <4 x double> [[TMP6]], <4 x double> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
 ; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, double [[C1]], i64 0
 ; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <4 x double> [[TMP8]], double [[C4]], i64 1
 ; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <4 x double> [[TMP9]], double [[C7]], i64 2
-; CHECK-NEXT:    [[TMP12:%.*]] = fmul reassoc nsz arcp contract afn <4 x double> [[TMP7]], [[TMP10]]
-; CHECK-NEXT:    [[TMP25:%.*]] = fadd reassoc nsz arcp contract afn <4 x double> [[TMP12]], [[TMP5]]
+; CHECK-NEXT:    [[TMP11:%.*]] = fmul fast <4 x double> [[TMP7]], [[TMP10]]
+; CHECK-NEXT:    [[TMP12:%.*]] = fadd fast <4 x double> [[TMP11]], [[TMP5]]
 ; CHECK-NEXT:    [[TMP13:%.*]] = insertelement <4 x double> <double poison, double -0.000000e+00, double poison, double poison>, double [[L2]], i64 0
 ; CHECK-NEXT:    [[TMP14:%.*]] = shufflevector <4 x double> [[TMP13]], <4 x double> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
 ; CHECK-NEXT:    [[TMP15:%.*]] = insertelement <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, double [[C2]], i64 0
 ; CHECK-NEXT:    [[TMP16:%.*]] = insertelement <4 x double> [[TMP15]], double [[C5]], i64 1
 ; CHECK-NEXT:    [[TMP17:%.*]] = insertelement <4 x double> [[TMP16]], double [[C8]], i64 2
 ; CHECK-NEXT:    [[TMP18:%.*]] = fmul fast <4 x double> [[TMP14]], [[TMP17]]
-; CHECK-NEXT:    [[TMP19:%.*]] = fadd reassoc nsz arcp contract afn <4 x double> [[TMP25]], [[TMP18]]
+; CHECK-NEXT:    [[TMP19:%.*]] = fadd fast <4 x double> [[TMP12]], [[TMP18]]
 ; CHECK-NEXT:    [[TMP20:%.*]] = shufflevector <4 x double> [[TMP19]], <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, <4 x i32> <i32 0, i32 1, i32 poison, i32 7>
 ; CHECK-NEXT:    [[TMP21:%.*]] = shufflevector <4 x double> [[TMP20]], <4 x double> [[TMP19]], <4 x i32> <i32 0, i32 1, i32 6, i32 3>
 ; CHECK-NEXT:    [[TMP22:%.*]] = fmul <4 x double> [[TMP19]], [[TMP21]]
-; CHECK-NEXT:    [[TMP24]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP22]])
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP22]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP23:%.*]] = phi double [ [[TMP24]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP23:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
 ; CHECK-NEXT:    ret double [[TMP23]]
 ;
 entry:
@@ -511,7 +511,7 @@ define double @two_exit_phis(ptr %p, i64 %n, i1 %c) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[TMP25:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[IV3:%.*]] = mul nuw i64 [[IV]], 3
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV3]]
 ; CHECK-NEXT:    [[P1:%.*]] = getelementptr double, ptr [[P0]], i64 1
@@ -534,31 +534,31 @@ define double @two_exit_phis(ptr %p, i64 %n, i1 %c) {
 ; CHECK-NEXT:    [[TMP3:%.*]] = insertelement <4 x double> [[TMP2]], double [[C3]], i64 1
 ; CHECK-NEXT:    [[TMP4:%.*]] = insertelement <4 x double> [[TMP3]], double [[C6]], i64 2
 ; CHECK-NEXT:    [[TMP5:%.*]] = fmul fast <4 x double> [[TMP1]], [[TMP4]]
-; CHECK-NEXT:    [[TMP11:%.*]] = insertelement <4 x double> poison, double [[L1]], i64 0
-; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <4 x double> [[TMP11]], double [[ACC]], i64 1
+; CHECK-NEXT:    [[TMP6:%.*]] = insertelement <4 x double> <double poison, double 0.000000e+00, double poison, double poison>, double [[L1]], i64 0
 ; CHECK-NEXT:    [[TMP7:%.*]] = shufflevector <4 x double> [[TMP6]], <4 x double> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
 ; CHECK-NEXT:    [[TMP8:%.*]] = insertelement <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, double [[C1]], i64 0
 ; CHECK-NEXT:    [[TMP9:%.*]] = insertelement <4 x double> [[TMP8]], double [[C4]], i64 1
 ; CHECK-NEXT:    [[TMP10:%.*]] = insertelement <4 x double> [[TMP9]], double [[C7]], i64 2
-; CHECK-NEXT:    [[TMP12:%.*]] = fmul reassoc nsz arcp contract afn <4 x double> [[TMP7]], [[TMP10]]
-; CHECK-NEXT:    [[TMP26:%.*]] = fadd reassoc nsz arcp contract afn <4 x double> [[TMP12]], [[TMP5]]
+; CHECK-NEXT:    [[TMP11:%.*]] = fmul fast <4 x double> [[TMP7]], [[TMP10]]
+; CHECK-NEXT:    [[TMP12:%.*]] = fadd fast <4 x double> [[TMP11]], [[TMP5]]
 ; CHECK-NEXT:    [[TMP13:%.*]] = insertelement <4 x double> <double poison, double -0.000000e+00, double poison, double poison>, double [[L2]], i64 0
 ; CHECK-NEXT:    [[TMP14:%.*]] = shufflevector <4 x double> [[TMP13]], <4 x double> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
 ; CHECK-NEXT:    [[TMP15:%.*]] = insertelement <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, double [[C2]], i64 0
 ; CHECK-NEXT:    [[TMP16:%.*]] = insertelement <4 x double> [[TMP15]], double [[C5]], i64 1
 ; CHECK-NEXT:    [[TMP17:%.*]] = insertelement <4 x double> [[TMP16]], double [[C8]], i64 2
 ; CHECK-NEXT:    [[TMP18:%.*]] = fmul fast <4 x double> [[TMP14]], [[TMP17]]
-; CHECK-NEXT:    [[TMP19:%.*]] = fadd reassoc nsz arcp contract afn <4 x double> [[TMP26]], [[TMP18]]
+; CHECK-NEXT:    [[TMP19:%.*]] = fadd fast <4 x double> [[TMP12]], [[TMP18]]
 ; CHECK-NEXT:    [[TMP20:%.*]] = shufflevector <4 x double> [[TMP19]], <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, <4 x i32> <i32 0, i32 1, i32 poison, i32 7>
 ; CHECK-NEXT:    [[TMP21:%.*]] = shufflevector <4 x double> [[TMP20]], <4 x double> [[TMP19]], <4 x i32> <i32 0, i32 1, i32 6, i32 3>
 ; CHECK-NEXT:    [[TMP22:%.*]] = fmul <4 x double> [[TMP19]], [[TMP21]]
-; CHECK-NEXT:    [[TMP25]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP22]])
+; CHECK-NEXT:    [[TMP25:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP22]])
+; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP25]], [[ACC]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP23:%.*]] = phi double [ [[TMP25]], %[[LOOP]] ]
-; CHECK-NEXT:    [[TMP24:%.*]] = phi double [ [[TMP25]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP23:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP24:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[R:%.*]] = fadd double [[TMP23]], [[TMP24]]
 ; CHECK-NEXT:    ret double [[R]]
 ;
@@ -622,21 +622,22 @@ define double @dup_preheader_edges(ptr %p, i64 %n, double %init, i32 %sw) {
 ; CHECK-LABEL: define double @dup_preheader_edges(
 ; CHECK-SAME: ptr [[P:%.*]], i64 [[N:%.*]], double [[INIT:%.*]], i32 [[SW:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[SLPRDX_INIT:%.*]] = insertelement <4 x double> zeroinitializer, double [[INIT]], i32 0
 ; CHECK-NEXT:    switch i32 [[SW]], label %[[LOOP:.*]] [
 ; CHECK-NEXT:      i32 1, label %[[LOOP]]
 ; CHECK-NEXT:    ]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ [[INIT]], %[[ENTRY]] ], [ [[INIT]], %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ [[SLPRDX_INIT]], %[[ENTRY]] ], [ [[SLPRDX_INIT]], %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
 ; CHECK-NEXT:    ret double [[TMP1]]
 ;
 entry:
@@ -679,16 +680,19 @@ define double @dup_exit_edges_bypass(ptr %p, i64 %n, double %y, i32 %sw) {
 ; CHECK-NEXT:    ]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ], [ poison, %[[ENTRY]] ], [ poison, %[[ENTRY]] ]
+; CHECK-NEXT:    [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LOOP]] ], [ false, %[[ENTRY]] ], [ false, %[[ENTRY]] ]
+; CHECK-NEXT:    [[RES:%.*]] = phi double [ poison, %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT:    [[TMP1:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP2]], double [[RES]]
 ; CHECK-NEXT:    ret double [[TMP1]]
 ;
 entry:
@@ -777,11 +781,10 @@ define double @early_exit_in_loop_value(ptr %p, i64 %n) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LATCH]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LATCH]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP3]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x double> [[TMP0]], i64 0
 ; CHECK-NEXT:    [[EC:%.*]] = fcmp ogt double [[TMP1]], 1.000000e+10
 ; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT:.*]], label %[[LATCH]]
@@ -790,7 +793,11 @@ define double @early_exit_in_loop_value(ptr %p, i64 %n) {
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP2:%.*]] = phi double [ [[OP_RDX]], %[[LATCH]] ], [ [[TMP1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LATCH]] ], [ poison, %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LATCH]] ], [ false, %[[LOOP]] ]
+; CHECK-NEXT:    [[RES:%.*]] = phi double [ poison, %[[LATCH]] ], [ [[TMP1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT:    [[TMP2:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP3]], double [[RES]]
 ; CHECK-NEXT:    ret double [[TMP2]]
 ;
 entry:
@@ -1295,11 +1302,10 @@ define double @bypass_from_sibling_loop(ptr %p, ptr %q, i64 %n, i64 %m, i1 %c) {
 ; CHECK-NEXT:    br i1 [[C]], label %[[LOOP:.*]], label %[[LOOP2:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP1]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
@@ -1313,7 +1319,11 @@ define double @bypass_from_sibling_loop(ptr %p, ptr %q, i64 %n, i64 %m, i1 %c) {
 ; CHECK-NEXT:    [[CMP2:%.*]] = icmp eq i64 [[J_NEXT]], [[M]]
 ; CHECK-NEXT:    br i1 [[CMP2]], label %[[EXIT]], label %[[LOOP2]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[RES:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ], [ [[SUM2]], %[[LOOP2]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ], [ poison, %[[LOOP2]] ]
+; CHECK-NEXT:    [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LOOP]] ], [ false, %[[LOOP2]] ]
+; CHECK-NEXT:    [[RES1:%.*]] = phi double [ poison, %[[LOOP]] ], [ [[SUM2]], %[[LOOP2]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT:    [[RES:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP1]], double [[RES1]]
 ; CHECK-NEXT:    ret double [[RES]]
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/loop-accumulator-reduction.ll b/llvm/test/Transforms/SLPVectorizer/X86/loop-accumulator-reduction.ll
index f05cc887d73154..55a31632227eb5 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/loop-accumulator-reduction.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/loop-accumulator-reduction.ll
@@ -16,16 +16,16 @@ define double @loop_acc_fadd(ptr %p) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
 ; CHECK-NEXT:    ret double [[TMP1]]
 ;
 entry:
@@ -217,17 +217,18 @@ define double @two_exit_phis(ptr %p) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP3]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
-; CHECK-NEXT:    [[TMP2:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT2:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT2]])
+; CHECK-NEXT:    [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
 ; CHECK-NEXT:    [[R:%.*]] = fadd double [[TMP1]], [[TMP2]]
 ; CHECK-NEXT:    ret double [[R]]
 ;
@@ -270,16 +271,16 @@ define double @loop_acc_fadd_unknown_trip_count(ptr %p, i64 %n) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
 ; CHECK-NEXT:    ret double [[TMP1]]
 ;
 entry:
@@ -319,16 +320,16 @@ define double @loop_acc_fmax(ptr %p, i64 %n) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[TMP3:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ <double 0.000000e+00, double -inf, double -inf, double -inf>, %[[ENTRY]] ], [ [[TMP1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = call nnan double @llvm.vector.reduce.fmax.v4f64(<4 x double> [[TMP0]])
-; CHECK-NEXT:    [[TMP3]] = call nnan double @llvm.maxnum.f64(double [[TMP1]], double [[ACC]])
+; CHECK-NEXT:    [[TMP1]] = call nnan <4 x double> @llvm.maxnum.v4f64(<4 x double> [[SLPRDX_ACC]], <4 x double> [[TMP0]])
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP2:%.*]] = phi double [ [[TMP3]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[TMP1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = call nnan double @llvm.vector.reduce.fmax.v4f64(<4 x double> [[SLPRDX_EXIT]])
 ; CHECK-NEXT:    ret double [[TMP2]]
 ;
 entry:
@@ -365,21 +366,22 @@ define double @dup_preheader_edges(ptr %p, i64 %n, double %init, i32 %sw) {
 ; CHECK-LABEL: define double @dup_preheader_edges(
 ; CHECK-SAME: ptr [[P:%.*]], i64 [[N:%.*]], double [[INIT:%.*]], i32 [[SW:%.*]]) #[[ATTR0]] {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[SLPRDX_INIT:%.*]] = insertelement <4 x double> zeroinitializer, double [[INIT]], i32 0
 ; CHECK-NEXT:    switch i32 [[SW]], label %[[LOOP:.*]] [
 ; CHECK-NEXT:      i32 1, label %[[LOOP]]
 ; CHECK-NEXT:    ]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ [[INIT]], %[[ENTRY]] ], [ [[INIT]], %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ [[SLPRDX_INIT]], %[[ENTRY]] ], [ [[SLPRDX_INIT]], %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
 ; CHECK-NEXT:    ret double [[TMP1]]
 ;
 entry:
@@ -423,16 +425,19 @@ define double @dup_exit_edges_bypass(ptr %p, i64 %n, double %y, i32 %sw) {
 ; CHECK-NEXT:    ]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP1]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[SLPRDX_SEL:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ], [ poison, %[[ENTRY]] ], [ poison, %[[ENTRY]] ]
+; CHECK-NEXT:    [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LOOP]] ], [ false, %[[ENTRY]] ], [ false, %[[ENTRY]] ]
+; CHECK-NEXT:    [[RES:%.*]] = phi double [ poison, %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT:    [[SLPRDX_SEL:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP1]], double [[RES]]
 ; CHECK-NEXT:    ret double [[SLPRDX_SEL]]
 ;
 entry:
@@ -474,16 +479,19 @@ define double @fmax_bypass(ptr %p, i64 %n, double %y, i1 %c) {
 ; CHECK-NEXT:    br i1 [[C]], label %[[LOOP:.*]], label %[[EXIT:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[TMP2:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ <double 0.000000e+00, double f0xFFEFFFFFFFFFFFFF, double f0xFFEFFFFFFFFFFFFF, double f0xFFEFFFFFFFFFFFFF>, %[[ENTRY]] ], [ [[TMP1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = call nnan ninf double @llvm.vector.reduce.fmax.v4f64(<4 x double> [[TMP0]])
-; CHECK-NEXT:    [[TMP2]] = call nnan ninf double @llvm.maxnum.f64(double [[TMP1]], double [[ACC]])
+; CHECK-NEXT:    [[TMP1]] = call nnan ninf <4 x double> @llvm.maxnum.v4f64(<4 x double> [[SLPRDX_ACC]], <4 x double> [[TMP0]])
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[SLPRDX_SEL:%.*]] = phi double [ [[TMP2]], %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[TMP1]], %[[LOOP]] ], [ poison, %[[ENTRY]] ]
+; CHECK-NEXT:    [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LOOP]] ], [ false, %[[ENTRY]] ]
+; CHECK-NEXT:    [[RES:%.*]] = phi double [ poison, %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = call nnan ninf double @llvm.vector.reduce.fmax.v4f64(<4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT:    [[SLPRDX_SEL:%.*]] = select nnan ninf i1 [[SLPRDX_FROMLOOP]], double [[TMP2]], double [[RES]]
 ; CHECK-NEXT:    ret double [[SLPRDX_SEL]]
 ;
 entry:
@@ -523,11 +531,10 @@ define double @early_exit_in_loop_value(ptr %p, i64 %n) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LATCH]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LATCH]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[TMP1:%.*]] = extractelement <4 x double> [[TMP0]], i64 0
 ; CHECK-NEXT:    [[EC:%.*]] = fcmp ogt double [[TMP1]], 1.000000e+10
 ; CHECK-NEXT:    br i1 [[EC]], label %[[EXIT:.*]], label %[[LATCH]]
@@ -536,7 +543,11 @@ define double @early_exit_in_loop_value(ptr %p, i64 %n) {
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[SLPRDX_SEL:%.*]] = phi double [ [[OP_RDX]], %[[LATCH]] ], [ [[TMP1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LATCH]] ], [ poison, %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LATCH]] ], [ false, %[[LOOP]] ]
+; CHECK-NEXT:    [[RES:%.*]] = phi double [ poison, %[[LATCH]] ], [ [[TMP1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT:    [[SLPRDX_SEL:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP2]], double [[RES]]
 ; CHECK-NEXT:    ret double [[SLPRDX_SEL]]
 ;
 entry:
@@ -645,14 +656,14 @@ define double @rdx_op_outside_loop(ptr %p, i64 %n) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
 ; CHECK-NEXT:    ret double [[TMP1]]
 ;
 entry:
@@ -1050,11 +1061,10 @@ define double @bypass_from_sibling_loop(ptr %p, ptr %q, i64 %n, i64 %m, i1 %c) {
 ; CHECK-NEXT:    br i1 [[C]], label %[[LOOP:.*]], label %[[LOOP2:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP1]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
@@ -1068,7 +1078,11 @@ define double @bypass_from_sibling_loop(ptr %p, ptr %q, i64 %n, i64 %m, i1 %c) {
 ; CHECK-NEXT:    [[CMP2:%.*]] = icmp eq i64 [[J_NEXT]], [[M]]
 ; CHECK-NEXT:    br i1 [[CMP2]], label %[[EXIT]], label %[[LOOP2]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[SLPRDX_SEL:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ], [ [[SUM2]], %[[LOOP2]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ], [ poison, %[[LOOP2]] ]
+; CHECK-NEXT:    [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LOOP]] ], [ false, %[[LOOP2]] ]
+; CHECK-NEXT:    [[RES:%.*]] = phi double [ poison, %[[LOOP]] ], [ [[SUM2]], %[[LOOP2]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT:    [[SLPRDX_SEL:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP1]], double [[RES]]
 ; CHECK-NEXT:    ret double [[SLPRDX_SEL]]
 ;
 entry:
@@ -1117,19 +1131,19 @@ define double @dot_product(ptr %p, ptr %q) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[Q0:%.*]] = getelementptr double, ptr [[Q]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
 ; CHECK-NEXT:    [[TMP1:%.*]] = load <4 x double>, ptr [[Q0]], align 8
 ; CHECK-NEXT:    [[TMP2:%.*]] = fmul fast <4 x double> [[TMP0]], [[TMP1]]
-; CHECK-NEXT:    [[TMP4:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP2]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP4]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP2]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP3:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
 ; CHECK-NEXT:    ret double [[TMP3]]
 ;
 entry:
@@ -1232,17 +1246,17 @@ define double @call_in_loop_fp(ptr %p) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT:    [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT:    [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
 ; CHECK-NEXT:    call void @sink()
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
 ; CHECK-NEXT:    ret double [[TMP1]]
 ;
 entry:
@@ -1282,17 +1296,17 @@ define i32 @zext_leaves(ptr %p) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x i32> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[P0:%.*]] = getelementptr i16, ptr [[P]], i64 [[IV]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = load <4 x i16>, ptr [[P0]], align 2
 ; CHECK-NEXT:    [[TMP1:%.*]] = zext <4 x i16> [[TMP0]] to <4 x i32>
-; CHECK-NEXT:    [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP1]])
-; CHECK-NEXT:    [[OP_RDX]] = add i32 [[TMP3]], [[ACC]]
+; CHECK-NEXT:    [[SLPRDX_ACC1]] = add <4 x i32> [[SLPRDX_ACC]], [[TMP1]]
 ; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
 ; CHECK-NEXT:    [[CMP:%.*]] = icmp eq i64 [[IV]], 64
 ; CHECK-NEXT:    br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[TMP2:%.*]] = phi i32 [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT:    [[SLPRDX_EXIT:%.*]] = phi <4 x i32> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[SLPRDX_EXIT]])
 ; CHECK-NEXT:    ret i32 [[TMP2]]
 ;
 entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/slp-fma-loss.ll b/llvm/test/Transforms/SLPVectorizer/X86/slp-fma-loss.ll
index f48748dd137b6e..1a7b3a9ff170cd 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/slp-fma-loss.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/slp-fma-loss.ll
@@ -11,12 +11,11 @@ define void @hr() {
 ; SSE4-LABEL: @hr(
 ; SSE4-NEXT:    br label [[LOOP:%.*]]
 ; SSE4:       loop:
-; SSE4-NEXT:    [[PHI0:%.*]] = phi double [ 0.000000e+00, [[TMP0:%.*]] ], [ [[OP_RDX:%.*]], [[LOOP]] ]
+; SSE4-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, [[TMP0:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[LOOP]] ]
 ; SSE4-NEXT:    [[CVT0:%.*]] = uitofp i16 0 to double
 ; SSE4-NEXT:    [[TMP1:%.*]] = insertelement <4 x double> <double poison, double 0.000000e+00, double 0.000000e+00, double 0.000000e+00>, double [[CVT0]], i64 0
 ; SSE4-NEXT:    [[TMP2:%.*]] = fmul fast <4 x double> zeroinitializer, [[TMP1]]
-; SSE4-NEXT:    [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP2]])
-; SSE4-NEXT:    [[OP_RDX]] = fadd fast double [[TMP3]], [[PHI0]]
+; SSE4-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP2]]
 ; SSE4-NEXT:    br i1 true, label [[EXIT:%.*]], label [[LOOP]]
 ; SSE4:       exit:
 ; SSE4-NEXT:    ret void
@@ -24,13 +23,11 @@ define void @hr() {
 ; AVX-LABEL: @hr(
 ; AVX-NEXT:    br label [[LOOP:%.*]]
 ; AVX:       loop:
-; AVX-NEXT:    [[PHI0:%.*]] = phi double [ 0.000000e+00, [[TMP0:%.*]] ], [ [[ADD3:%.*]], [[LOOP]] ]
+; AVX-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, [[TMP0:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[LOOP]] ]
 ; AVX-NEXT:    [[CVT0:%.*]] = uitofp i16 0 to double
-; AVX-NEXT:    [[MUL0:%.*]] = fmul fast double 0.000000e+00, [[CVT0]]
-; AVX-NEXT:    [[ADD0:%.*]] = fadd fast double [[MUL0]], [[PHI0]]
-; AVX-NEXT:    [[ADD1:%.*]] = fadd fast double 0.000000e+00, [[ADD0]]
-; AVX-NEXT:    [[ADD2:%.*]] = fadd fast double 0.000000e+00, [[ADD1]]
-; AVX-NEXT:    [[ADD3]] = fadd fast double 0.000000e+00, [[ADD2]]
+; AVX-NEXT:    [[TMP1:%.*]] = insertelement <4 x double> <double poison, double 0.000000e+00, double 0.000000e+00, double 0.000000e+00>, double [[CVT0]], i64 0
+; AVX-NEXT:    [[TMP2:%.*]] = fmul fast <4 x double> zeroinitializer, [[TMP1]]
+; AVX-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP2]]
 ; AVX-NEXT:    br i1 true, label [[EXIT:%.*]], label [[LOOP]]
 ; AVX:       exit:
 ; AVX-NEXT:    ret void
@@ -38,12 +35,11 @@ define void @hr() {
 ; AVX2-LABEL: @hr(
 ; AVX2-NEXT:    br label [[LOOP:%.*]]
 ; AVX2:       loop:
-; AVX2-NEXT:    [[PHI0:%.*]] = phi double [ 0.000000e+00, [[TMP0:%.*]] ], [ [[OP_RDX:%.*]], [[LOOP]] ]
+; AVX2-NEXT:    [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, [[TMP0:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[LOOP]] ]
 ; AVX2-NEXT:    [[CVT0:%.*]] = uitofp i16 0 to double
 ; AVX2-NEXT:    [[TMP1:%.*]] = insertelement <4 x double> <double poison, double 0.000000e+00, double 0.000000e+00, double 0.000000e+00>, double [[CVT0]], i64 0
 ; AVX2-NEXT:    [[TMP2:%.*]] = fmul fast <4 x double> zeroinitializer, [[TMP1]]
-; AVX2-NEXT:    [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP2]])
-; AVX2-NEXT:    [[OP_RDX]] = fadd fast double [[TMP3]], [[PHI0]]
+; AVX2-NEXT:    [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP2]]
 ; AVX2-NEXT:    br i1 true, label [[EXIT:%.*]], label [[LOOP]]
 ; AVX2:       exit:
 ; AVX2-NEXT:    ret void



More information about the llvm-commits mailing list