[llvm] [SLP]Emit loop-carried horizontal reductions as loop vector accumulator (PR #221598)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 7 11:24:29 PDT 2026
https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/221598
>From 046466febcb6cbd39aa51bfec96b56b2b9fff134 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Sun, 6 Sep 2026 12:16:44 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 588 +++++++++++++++++-
.../SLPVectorizer/AArch64/gather-root.ll | 15 +-
.../SLPVectorizer/AArch64/getelementptr.ll | 42 +-
.../SLPVectorizer/AArch64/horizontal.ll | 19 +-
.../AArch64/loop-accumulator-reduction.ll | 76 ++-
.../X86/loop-accumulator-reduction.ll | 120 ++--
.../SLPVectorizer/X86/slp-fma-loss.ll | 20 +-
7 files changed, 726 insertions(+), 154 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 6d1241fe4408f..e4ef237f2ecd6 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -145,6 +145,11 @@ static cl::opt<bool>
ShouldVectorizeHor("slp-vectorize-hor", cl::init(true), cl::Hidden,
cl::desc("Attempt to vectorize horizontal reductions"));
+static cl::opt<bool> VectorizeLoopAccRdx(
+ "slp-vectorize-loop-acc-rdx", cl::init(true), cl::Hidden,
+ cl::desc("Emit loop-carried horizontal reductions as a vector "
+ "accumulator in the loop plus a single reduction after it"));
+
static cl::opt<bool> ShouldStartVectorizeHorAtStore(
"slp-vectorize-hor-store", cl::init(false), cl::Hidden,
cl::desc(
@@ -489,6 +494,10 @@ getNumberOfParts(const TargetTransformInfo &TTI, Type *VecTy, Type *ScalarTy,
return NumParts;
}
+namespace {
+class HorizontalReduction;
+} // namespace
+
/// Bottom Up SLP Vectorizer.
class slpvectorizer::BoUpSLP {
class TreeEntry;
@@ -4343,6 +4352,7 @@ class slpvectorizer::BoUpSLP {
friend struct GraphTraits<BoUpSLP *>;
friend struct DOTGraphTraits<BoUpSLP *>;
+ friend class ::HorizontalReduction;
/// Contains all scheduling data for a basic block.
/// It does not schedules instructions, which are not memory read/write
@@ -31052,10 +31062,497 @@ class HorizontalReduction {
return true;
}
+ /// Loop accumulator emission mode of the reduction: a reduction that
+ /// accumulates a loop phi (the phi is one of the reduced values and the
+ /// root is its backedge value) can be emitted as a vector accumulator in
+ /// the loop plus a single reduction after the loop instead of a horizontal
+ /// reduction on every iteration.
+ class LoopAccumulator {
+ /// The reduction root and the reduction operation kind and ordering.
+ Instruction *Root = nullptr;
+ RecurKind RdxKind = RecurKind::None;
+ ReductionOrdering RK = ReductionOrdering::None;
+ /// The vectorizer state and the analyses the accumulator cost and
+ /// emission are computed with.
+ BoUpSLP &R;
+ const TargetTransformInfo &TTI;
+ LoopInfo &LI;
+ /// The accumulated phi and the identity constant that replaced it in the
+ /// reduced values. The identity is never folded as a leftover: it does
+ /// not change the result.
+ PHINode *Phi = nullptr;
+ Constant *Identity = nullptr;
+ /// The number of calls in the loop, across which the vector accumulator
+ /// is kept live.
+ unsigned NumCalls = 0;
+ /// Exit phis consuming the reduction result outside the loop.
+ SmallSetVector<PHINode *, 2> ExitPhis;
+
+ /// \returns the identity constant of the reduction operation for \p Ty
+ /// under the fast-math flags \p FMF.
+ Constant *getIdentity(Type *Ty, FastMathFlags FMF) const {
+ Intrinsic::ID Id;
+ switch (RdxKind) {
+ case RecurKind::FMax:
+ case RecurKind::FMaxNum:
+ Id = Intrinsic::vector_reduce_fmax;
+ break;
+ case RecurKind::FMin:
+ case RecurKind::FMinNum:
+ Id = Intrinsic::vector_reduce_fmin;
+ break;
+ case RecurKind::FMaximum:
+ Id = Intrinsic::vector_reduce_fmaximum;
+ break;
+ case RecurKind::FMinimum:
+ Id = Intrinsic::vector_reduce_fminimum;
+ break;
+ default:
+ Id = RecurrenceDescriptor::isIntMinMaxRecurrenceKind(RdxKind)
+ ? getMinMaxReductionIntrinsicID(
+ getMinMaxReductionIntrinsicOp(RdxKind))
+ : getReductionForBinop(static_cast<Instruction::BinaryOps>(
+ RecurrenceDescriptor::getOpcode(RdxKind)));
+ break;
+ }
+ return cast<Constant>(getReductionIdentity(Id, Ty, FMF));
+ }
+
+ /// \returns the extra cost of accumulating the slice lane-wise into the
+ /// vector accumulator instead of reducing it on every iteration: the
+ /// lane-wise operation, the spill and reload on every iteration of the
+ /// values live beyond the register file and keeping the vector
+ /// accumulator live across the calls of the loop.
+ InstructionCost getAccumulationCost(FastMathFlags FMF, VectorType *VectorTy,
+ Type *ScalarTy) const {
+ const TTI::TargetCostKind CostKind = R.getCostKind();
+ auto GetOpCost = [&](Type *OpTy) {
+ if (RecurrenceDescriptor::isMinMaxRecurrenceKind(RdxKind)) {
+ IntrinsicCostAttributes ICA(getMinMaxReductionIntrinsicOp(RdxKind),
+ OpTy, {OpTy, OpTy}, FMF);
+ return TTI.getIntrinsicInstrCost(ICA, CostKind);
+ }
+ return TTI.getArithmeticInstrCost(
+ RecurrenceDescriptor::getOpcode(RdxKind), OpTy, CostKind);
+ };
+ // The lane-wise operation; also replaces the scalar operation folding
+ // the accumulator phi on top of the horizontal reduction (the scalar
+ // cost of the slice credits it neither, the slice lacks the phi).
+ InstructionCost Cost = GetOpCost(VectorTy) - GetOpCost(ScalarTy);
+ // The vector accumulator and the reduced vector are live at the same
+ // time; the registers they need beyond the register file are spilled
+ // and reloaded on every iteration.
+ constexpr unsigned NumLiveVectors = 2;
+ unsigned Parts = R.getNumberOfParts(VectorTy, ScalarTy);
+ unsigned RC = TTI.getRegisterClassForType(/*Vector=*/true, VectorTy);
+ unsigned NumRegs = TTI.getNumberOfRegisters(RC);
+ InstructionCost SpillReload =
+ TTI.getRegisterClassSpillCost(RC, CostKind) +
+ TTI.getRegisterClassReloadCost(RC, CostKind);
+ if (NumRegs != 0 && NumLiveVectors * Parts > NumRegs)
+ Cost += SpillReload * (NumLiveVectors * Parts - NumRegs);
+ // The vector accumulator replaces the scalar one across the calls of
+ // the loop: without callee-saved vector registers it is spilled and
+ // reloaded around every call, like a scalar FP accumulator.
+ if (NumCalls != 0) {
+ InstructionCost VecLive = std::max(
+ TTI.getCostOfKeepingLiveOverCall(VectorTy), SpillReload * Parts);
+ InstructionCost ScalarLive = TTI.getCostOfKeepingLiveOverCall(ScalarTy);
+ if (ScalarTy->isFloatingPointTy()) {
+ unsigned SRC =
+ TTI.getRegisterClassForType(/*Vector=*/false, ScalarTy);
+ ScalarLive = std::max(
+ ScalarLive, TTI.getRegisterClassSpillCost(SRC, CostKind) +
+ TTI.getRegisterClassReloadCost(SRC, CostKind));
+ }
+ Cost += (VecLive - ScalarLive) * NumCalls;
+ }
+ return Cost;
+ }
+
+ public:
+ LoopAccumulator(Instruction *Root, RecurKind RdxKind, ReductionOrdering RK,
+ BoUpSLP &R, const TargetTransformInfo &TTI, LoopInfo &LI)
+ : Root(Root), RdxKind(RdxKind), RK(RK), R(R), TTI(TTI), LI(LI) {}
+
+ /// If the reduction accumulates a loop phi (the phi is one of the reduced
+ /// values and the root is its backedge value), replaces the phi by the
+ /// reduction identity in the reduced values, so that the emitted tree
+ /// does not reference the phi. The identity does not change the result,
+ /// so if the accumulator emission does not apply, the phi is just folded
+ /// back on top of the emitted reduction. All unordered reduction kinds
+ /// have an identity constant.
+ void prepare(FastMathFlags RdxFMF,
+ SmallVectorImpl<SmallVector<Value *>> &ReducedVals,
+ SmallDenseMap<Value *, SmallVector<Instruction *>, 16>
+ &ReducedValsToOps,
+ bool HasNarrowedLeafShifts) {
+ if (!VectorizeLoopAccRdx || RK != ReductionOrdering::Unordered ||
+ Root->getType()->isIntegerTy(1) || Root->getType()->isVectorTy() ||
+ HasNarrowedLeafShifts || isCmpSelMinMax(Root) ||
+ Root->hasNUsesOrMore(UsesLimit))
+ return;
+ Loop *L = LI.getLoopFor(Root->getParent());
+ if (!L)
+ return;
+ BasicBlock *Latch = L->getLoopLatch();
+ if (!Latch)
+ return;
+ // Only a single accumulated phi, used by the reduction operations
+ // only, is supported.
+ auto FindAccPhi = [&]() -> PHINode * {
+ PHINode *AccPhi = nullptr;
+ for (ArrayRef<Value *> Candidates : ReducedVals)
+ for (Value *RdxVal : Candidates) {
+ auto *P = dyn_cast<PHINode>(RdxVal);
+ if (!P || P->getParent() != L->getHeader() ||
+ P->getNumIncomingValues() > MaxPHINumOperands ||
+ P->getIncomingValueForBlock(Latch) != Root)
+ continue;
+ if (AccPhi || P->hasNUsesOrMore(UsesLimit) ||
+ P->getNumUses() != ReducedValsToOps.at(P).size())
+ return nullptr;
+ AccPhi = P;
+ }
+ return AccPhi;
+ };
+ PHINode *AccPhi = FindAccPhi();
+ if (!AccPhi)
+ return;
+ Constant *IdC = getIdentity(AccPhi->getType(), RdxFMF);
+ if (ReducedValsToOps.contains(IdC))
+ return;
+ // The root may be used only by the accumulator phi and by phis outside
+ // the loop (the exit phis). The initial values are inserted into the
+ // identity vector at the end of their incoming blocks, so they must not
+ // be defined by the terminators; the final reduction is emitted at the
+ // beginning of the exit blocks, which must not be EH pads. Neither may
+ // land in a loop not containing the reduction loop, which would execute
+ // them repeatedly.
+ auto IsExecutedOncePerLoop = [&](BasicBlock *BB) {
+ Loop *BBL = LI.getLoopFor(BB);
+ return !BBL || BBL->contains(L);
+ };
+ if (any_of(zip(AccPhi->incoming_values(), AccPhi->blocks()),
+ [&](const auto &P) {
+ auto [IncV, B] = P;
+ return IncV != Root && (IncV == B->getTerminator() ||
+ !IsExecutedOncePerLoop(B));
+ }))
+ return;
+ auto IsValidExitPhi = [&](User *U) {
+ auto *ExitPhi = dyn_cast<PHINode>(U);
+ return ExitPhi &&
+ ExitPhi->getNumIncomingValues() <= MaxPHINumOperands &&
+ !ExitPhi->getParent()->isEHPad() &&
+ !L->contains(ExitPhi->getParent()) &&
+ IsExecutedOncePerLoop(ExitPhi->getParent());
+ };
+ auto CollectExitPhis = [&] {
+ for (User *U : Root->users()) {
+ if (U == AccPhi)
+ continue;
+ if (!IsValidExitPhi(U))
+ return false;
+ ExitPhis.insert(cast<PHINode>(U));
+ }
+ return true;
+ };
+ if (!CollectExitPhis())
+ return;
+ // The final reduction and the insert of the initial value would be on
+ // the loop-carried chain of an enclosing loop if the reduction result
+ // feeds the initial value of the accumulator through it: keep the
+ // horizontal reduction, whose chain is the scalar accumulator only.
+ auto IsFedByExitPhi = [&](Value *InitV) {
+ auto *InitPhi = dyn_cast<PHINode>(InitV);
+ // Too many incoming values to scan: keep the horizontal reduction.
+ if (InitV == Root || !InitPhi)
+ return false;
+ if (InitPhi->getNumIncomingValues() > MaxPHINumOperands)
+ return true;
+ return ExitPhis.contains(InitPhi) ||
+ any_of(InitPhi->incoming_values(), [&](Value *V) {
+ return isa<PHINode>(V) && ExitPhis.contains(cast<PHINode>(V));
+ });
+ };
+ if (any_of(AccPhi->incoming_values(), IsFedByExitPhi))
+ return;
+ // Calls clobber the vector registers: the vector accumulator has to be
+ // kept live across them. Intrinsics cheaper than a call are not calls.
+ for (BasicBlock *BB : L->blocks())
+ NumCalls += count_if(*BB, [&](const Instruction &I) {
+ auto *CB = dyn_cast<CallBase>(&I);
+ if (!CB || CB->doesNotReturn())
+ return false;
+ auto *II = dyn_cast<IntrinsicInst>(CB);
+ if (!II)
+ return true;
+ if (II->isAssumeLikeIntrinsic())
+ return false;
+ IntrinsicCostAttributes ICA(II->getIntrinsicID(), *II);
+ return TTI.getIntrinsicInstrCost(ICA, R.getCostKind()) >=
+ TTI.getCallInstrCost(nullptr, II->getType(), ICA.getArgTypes(),
+ R.getCostKind());
+ });
+ for (SmallVector<Value *> &Candidates : ReducedVals)
+ for (Value *&RdxVal : Candidates)
+ if (RdxVal == AccPhi)
+ RdxVal = IdC;
+ ReducedValsToOps.try_emplace(IdC, ReducedValsToOps.lookup(AccPhi));
+ ReducedValsToOps.erase(AccPhi);
+ Phi = AccPhi;
+ Identity = IdC;
+ }
+
+ /// Collects the reduced values that were not vectorized; they are folded
+ /// into the reduction on top of the vectorized part. The accumulator
+ /// identity constant is never folded.
+ void collectLeftovers(
+ const DenseMap<Value *, unsigned> &VectorizedVals,
+ ArrayRef<SmallVector<Value *>> ReducedVals,
+ const SmallDenseMap<Value *, SmallVector<Instruction *>, 16>
+ &ReducedValsToOps,
+ SmallVectorImpl<std::pair<Instruction *, Value *>> &LeftoverReductions)
+ const {
+ SmallPtrSet<Value *, 8> Visited;
+ for (ArrayRef<Value *> Candidates : ReducedVals)
+ for (Value *RdxVal : Candidates) {
+ if (RdxVal == Identity || !Visited.insert(RdxVal).second)
+ continue;
+ unsigned NumOps = VectorizedVals.lookup(RdxVal);
+ for (Instruction *RedOp :
+ ArrayRef(ReducedValsToOps.at(RdxVal)).drop_back(NumOps))
+ LeftoverReductions.emplace_back(RedOp, RdxVal);
+ }
+ }
+
+ /// \returns the cost of the code the vector accumulator emits outside the
+ /// loop: the final reductions in the exit blocks and the insert of the
+ /// initial value, scaled by the enclosing loop nest.
+ InstructionCost getExitCost(FastMathFlags FMF, const Loop *L) const {
+ assert(
+ (RecurrenceDescriptor::isMinMaxRecurrenceKind(RdxKind) ||
+ Instruction::isBinaryOp(RecurrenceDescriptor::getOpcode(RdxKind))) &&
+ "Expected arithmetic or min/max reduction operation");
+ FixedVectorType *VecTy = R.getReductionType();
+ TTI::TargetCostKind CostKind = R.getCostKind();
+ InstructionCost RdxCost =
+ RecurrenceDescriptor::isMinMaxRecurrenceKind(RdxKind)
+ ? TTI.getMinMaxReductionCost(
+ getMinMaxReductionIntrinsicOp(RdxKind), VecTy, FMF,
+ CostKind)
+ : TTI.getArithmeticReductionCost(
+ RecurrenceDescriptor::getOpcode(RdxKind), VecTy, FMF,
+ CostKind);
+ InstructionCost Cost =
+ RdxCost * ExitPhis.size() +
+ TTI.getVectorInstrCost(Instruction::InsertElement, VecTy, CostKind,
+ /*Index=*/0);
+ return Cost * R.getLoopNestScale(L->getParentLoop());
+ }
+
+ /// Checks that the slice can be emitted as the vector accumulator: it is
+ /// the only one (all other reduced values are the identity), not scaled,
+ /// not reduced in-tree, of the root type and fully vectorized. An integer
+ /// accumulator stays in a callee-saved register across the calls of the
+ /// loop, the vector one would be spilled around every call on the
+ /// loop-carried chain.
+ bool isCandidate(const Value *VectorizedTree, bool HasVectorizedSlices,
+ bool IsSupportedHorRdxIdentityOp,
+ ArrayRef<SmallVector<Value *>> ReducedVals,
+ ArrayRef<Value *> Candidates, unsigned I, unsigned Pos,
+ unsigned ReduxWidth, bool OptReusedScalars,
+ bool SameScaleFactor) const {
+ if (!Phi || VectorizedTree || HasVectorizedSlices || Pos != 0 ||
+ (OptReusedScalars && SameScaleFactor) || R.isReducedBitcastRoot() ||
+ R.isReducedCmpBitcastRoot() ||
+ (NumCalls != 0 && !Root->getType()->isFloatingPointTy()))
+ return false;
+ if (!all_of(ArrayRef(Candidates).drop_front(ReduxWidth),
+ [&](Value *RdxVal) { return RdxVal == Identity; }))
+ return false;
+ if (!all_of(enumerate(ReducedVals), [&](const auto &P) {
+ auto [Idx, RV] = P;
+ return Idx == I || (RV.size() == 1 && RV.front() == Identity);
+ }))
+ return false;
+ if (R.getReductionType()->getElementType() != Root->getType())
+ return false;
+ return IsSupportedHorRdxIdentityOp ||
+ all_of(Candidates.slice(Pos, ReduxWidth),
+ [&](Value *RdxVal) { return R.isVectorized(RdxVal); });
+ }
+
+ /// \returns the cost delta of the vector accumulator form vs the
+ /// horizontal reduction on every iteration. Both remove the same scalar
+ /// operations, so only the vector code is compared: the lane-wise
+ /// operation is executed on every iteration of the loop, the horizontal
+ /// reduction as often as the tree cost model assumes (once, if the
+ /// reduced values are loop-invariant), the final reduction and the
+ /// initial insert once, outside the loop.
+ InstructionCost getCostDelta(InstructionCost HorVecCost,
+ InstructionCost RdxOpCost, FastMathFlags FMF,
+ VectorType *VectorTy, Type *ScalarTy) const {
+ // The lane-wise accumulation replaces the reduction operation in the
+ // vector code.
+ InstructionCost AccVecCost =
+ HorVecCost - RdxOpCost + getAccumulationCost(FMF, VectorTy, ScalarTy);
+ Loop *L = LI.getLoopFor(Phi->getParent());
+ InstructionCost Delta = AccVecCost * R.getLoopNestScale(L) -
+ HorVecCost * R.getScaleToLoopIterations(
+ R.getRootNode(), nullptr, Root) +
+ getExitCost(FMF, L);
+ LLVM_DEBUG(dbgs() << "SLP: Loop accumulator cost delta " << Delta
+ << "\n");
+ return Delta;
+ }
+
+ /// Emits the reduction as a vector accumulator if the accumulator form
+ /// applies; otherwise folds the accumulator phi back on top of the
+ /// reduction like a leftover and returns null.
+ Value *tryEmit(
+ IRBuilderBase &Builder, FastMathFlags RdxFMF, bool Vectorized,
+ const Value *VectorizedTree,
+ SmallVectorImpl<std::pair<Instruction *, Value *>> &LeftoverReductions,
+ const SmallVector<std::tuple<WeakTrackingVH, unsigned, bool, bool>>
+ &VectorValuesAndScales,
+ const SmallPtrSetImpl<Value *> &RequiredExtract,
+ const SmallDenseMap<Value *, SmallVector<Instruction *>, 16>
+ &ReducedValsToOps,
+ const ReductionOpsListType &ReductionOps,
+ function_ref<Value *(Value *, IRBuilderBase &, Type *)> EmitReduction) {
+ if (!shouldEmit(Vectorized, VectorizedTree, LeftoverReductions,
+ VectorValuesAndScales, RequiredExtract)) {
+ // The accumulator phi is folded back on top of the reduction like a
+ // leftover: the identity replaced it in the reduced values and does
+ // not change the result.
+ if (Phi)
+ LeftoverReductions.emplace_back(ReducedValsToOps.at(Identity).front(),
+ Phi);
+ return nullptr;
+ }
+ return emit(Builder, RdxFMF, std::get<0>(VectorValuesAndScales.front()),
+ ReductionOps, EmitReduction);
+ }
+
+ private:
+ /// Checks that the accumulator form applies: the slice was vectorized as
+ /// the accumulator and no other vectorized value or leftover reduced
+ /// values remain to fold; the slice is of the root type, not scaled, not
+ /// reduced in-tree and without extracts.
+ bool shouldEmit(
+ bool Vectorized, const Value *VectorizedTree,
+ ArrayRef<std::pair<Instruction *, Value *>> LeftoverReductions,
+ const SmallVector<std::tuple<WeakTrackingVH, unsigned, bool, bool>>
+ &VectorValuesAndScales,
+ const SmallPtrSetImpl<Value *> &RequiredExtract) const {
+ if (!Vectorized || VectorizedTree || !LeftoverReductions.empty() ||
+ VectorValuesAndScales.size() != 1 || !RequiredExtract.empty())
+ return false;
+ const auto &[Vec, Scale, IsSigned, ReducedInTree] =
+ VectorValuesAndScales.front();
+ return Scale == 1 && !ReducedInTree &&
+ cast<VectorType>(Vec->getType())->getElementType() ==
+ Root->getType();
+ }
+
+ /// Emits the reduction as a vector accumulator: the vectorized reduced
+ /// values are accumulated lane-wise into a vector phi in the loop and
+ /// reduced once in the exit blocks.
+ Value *emit(
+ IRBuilderBase &Builder, FastMathFlags RdxFMF, Value *Vec,
+ const ReductionOpsListType &ReductionOps,
+ function_ref<Value *(Value *, IRBuilderBase &, Type *)> EmitReduction) {
+ auto *VecTy = cast<FixedVectorType>(Vec->getType());
+ // The initial values are placed in lane 0 of the identity splat at the
+ // end of their incoming blocks; multiple edges from a block carry the
+ // same value and share the insert.
+ Constant *IdVec =
+ ConstantVector::getSplat(VecTy->getElementCount(), Identity);
+ auto *VAcc = PHINode::Create(VecTy, Phi->getNumIncomingValues(),
+ "slprdx.acc", Phi->getIterator());
+ SmallDenseMap<BasicBlock *, Value *> InitVecs;
+ for (auto [IncV, B] : zip(Phi->incoming_values(), Phi->blocks())) {
+ if (IncV == Root)
+ continue;
+ Value *&InitVec = InitVecs[B];
+ if (!InitVec) {
+ IRBuilder<> EB(B->getTerminator());
+ InitVec = EB.CreateInsertElement(IdVec, IncV, EB.getInt32(0),
+ "slprdx.init");
+ }
+ VAcc->addIncoming(InitVec, B);
+ }
+ Builder.SetInsertPoint(Root);
+ Value *VAdd =
+ createOp(Builder, RdxKind, VAcc, Vec, "slprdx.acc", ReductionOps);
+ // The accumulation is on the loop-carried chain: do not let it be
+ // contracted into an FMA with the reduced multiplication, which would
+ // lengthen the chain.
+ if (auto *FPOp = dyn_cast<FPMathOperator>(VAdd)) {
+ FastMathFlags FMF = FPOp->getFastMathFlags();
+ FMF.setAllowContract(false);
+ cast<Instruction>(VAdd)->copyFastMathFlags(FMF);
+ }
+ // The accumulator is the backedge value.
+ for (auto [IncV, B] : zip(Phi->incoming_values(), Phi->blocks()))
+ if (IncV == Root)
+ VAcc->addIncoming(VAdd, B);
+ LLVM_DEBUG(dbgs() << "SLP: Emitting loop accumulator reduction for "
+ "reduction with root "
+ << *Root << "\n");
+ // The root is tracked by a weak handle: break its uses on the phi side.
+ Value *Poison = PoisonValue::get(Root->getType());
+ for (PHINode *ExitPhi : ExitPhis) {
+ // The accumulator replaces the root on the edges carrying it and is
+ // reduced once in the exit block. The values bypassing the loop never
+ // went through the reduction operations and must stay exact: the
+ // scalar phi keeps them and they are selected past the reduction.
+ unsigned NumIncoming = ExitPhi->getNumIncomingValues();
+ auto *VExit = PHINode::Create(VecTy, NumIncoming, "slprdx.exit",
+ ExitPhi->getIterator());
+ PHINode *FromLoop = nullptr;
+ if (any_of(ExitPhi->incoming_values(),
+ [this](Value *V) { return V != Root; }))
+ FromLoop = PHINode::Create(Builder.getInt1Ty(), NumIncoming,
+ "slprdx.fromloop", ExitPhi->getIterator());
+ Value *VecPoison = PoisonValue::get(VecTy);
+ for (auto [V, B] : zip(ExitPhi->incoming_values(), ExitPhi->blocks())) {
+ bool IsRoot = V == Root;
+ VExit->addIncoming(IsRoot ? VAdd : VecPoison, B);
+ if (FromLoop)
+ FromLoop->addIncoming(Builder.getInt1(IsRoot), B);
+ }
+ ExitPhi->replaceUsesOfWith(Root, Poison);
+ IRBuilder<> XB(&*ExitPhi->getParent()->getFirstInsertionPt());
+ XB.SetCurrentDebugLocation(ExitPhi->getDebugLoc());
+ XB.setFastMathFlags(RdxFMF);
+ Value *Res = EmitReduction(VExit, XB, ExitPhi->getType());
+ if (!FromLoop) {
+ ExitPhi->replaceAllUsesWith(Res);
+ // The erasure is deferred: the block iteration is still in progress.
+ R.eraseInstruction(ExitPhi);
+ continue;
+ }
+ Value *Sel = XB.CreateSelectWithUnknownProfile(
+ FromLoop, Res, ExitPhi, DEBUG_TYPE, "slprdx.sel");
+ ExitPhi->replaceUsesWithIf(
+ Sel, [Sel](Use &U) { return U.getUser() != Sel; });
+ }
+ Phi->replaceUsesOfWith(Root, Poison);
+ R.eraseInstruction(Phi);
+ // Do not let the caller re-analyze the emitted vector operation as a new
+ // reduction seed.
+ R.analyzedReductionRoot(cast<Instruction>(VAdd));
+ return VAdd;
+ }
+ };
+
/// Attempt to vectorize the tree found by matchAssociativeReduction.
Value *tryToReduce(BoUpSLP &V, const DataLayout &DL, TargetTransformInfo *TTI,
const TargetLibraryInfo &TLI, AssumptionCache *AC,
- DominatorTree &DT) {
+ DominatorTree &DT, LoopInfo &LI) {
constexpr unsigned RegMaxNumber = 4;
const unsigned RedValsMaxNumber =
(RK == ReductionOrdering::Ordered &&
@@ -31171,6 +31668,15 @@ class HorizontalReduction {
IgnoreList.clear();
bool IsCmpSelMinMax = isCmpSelMinMax(cast<Instruction>(ReductionRoot));
+ // A reduction that accumulates a loop phi can be emitted as a vector
+ // accumulator in the loop plus a single reduction after it.
+ LoopAccumulator LoopAcc(cast<Instruction>(ReductionRoot), RdxKind, RK, V,
+ *TTI, LI);
+ LoopAcc.prepare(RdxFMF, ReducedVals, ReducedValsToOps,
+ !NarrowedLeafShifts.empty());
+ // Set when the slice costed as the vector accumulator was vectorized.
+ bool LoopAccVectorized = false;
+
// Need to track reduced vals, they may be changed during vectorization of
// subvectors.
for (ArrayRef<Value *> Candidates : ReducedVals)
@@ -31542,6 +32048,10 @@ class HorizontalReduction {
V.transformNodes();
V.computeMinimumValueSizes();
InstructionCost TreeCost = V.calculateTreeCostAndTrimNonProfitable(VL);
+ const bool LoopAccCandidate = LoopAcc.isCandidate(
+ VectorizedTree, !VectorValuesAndScales.empty(),
+ IsSupportedHorRdxIdentityOp, ReducedVals, Candidates, I, Pos,
+ ReduxWidth, OptReusedScalars, SameScaleFactor);
SmallPtrSet<Value *, 4> VLScalars(llvm::from_range, VL);
// Gather externally used values.
@@ -31575,14 +32085,25 @@ class HorizontalReduction {
V.buildExternalUses(LocalExternallyUsedValues);
// Estimate cost.
- InstructionCost ReductionCost;
+ InstructionCost ReductionCost, HorVecCost = 0, RdxOpCost = 0;
if (RK == ReductionOrdering::Ordered || V.isReducedBitcastRoot() ||
V.isReducedCmpBitcastRoot())
ReductionCost = 0;
else
ReductionCost =
getReductionCost(TTI, VL, SameValuesCounter, IsCmpSelMinMax,
- RdxFMF, V, DT, DL, TLI);
+ RdxFMF, V, DT, DL, TLI, HorVecCost, RdxOpCost);
+ // The vector accumulator form is used if it is not more expensive than
+ // the horizontal reduction on every iteration.
+ InstructionCost LoopAccDelta = 0;
+ if (LoopAccCandidate && ReductionCost.isValid())
+ LoopAccDelta =
+ LoopAcc.getCostDelta(HorVecCost, RdxOpCost, RdxFMF,
+ V.getReductionType(), VL.front()->getType());
+ // On a tie the accumulator wins: it also replaces the loop-carried
+ // scalar dependency through the horizontal reduction by a lane-wise
+ // one.
+ const bool UseLoopAccForm = LoopAccCandidate && LoopAccDelta <= 0;
// If the root is a select (min/max idiom), the insert point is the
// compare condition of that select.
Instruction *RdxRootInst = cast<Instruction>(ReductionRoot);
@@ -31591,6 +32112,8 @@ class HorizontalReduction {
InsertPt = GetCmpForMinMaxReduction(RdxRootInst);
InstructionCost Cost =
V.getTreeCost(TreeCost, VL, ReductionCost, InsertPt);
+ if (UseLoopAccForm)
+ Cost += LoopAccDelta;
LLVM_DEBUG(dbgs() << "SLP: Found cost = " << Cost
<< " for reduction\n");
if (!Cost.isValid())
@@ -31729,6 +32252,7 @@ class HorizontalReduction {
? NarrowedLeafShifts.empty() && V.isSignedMinBitwidthRootNode()
: true,
V.isReducedBitcastRoot() || V.isReducedCmpBitcastRoot());
+ LoopAccVectorized = UseLoopAccForm;
// Count vectorized reduced values to exclude them from final reduction.
for (const auto [Idx, RdxVal] : enumerate(VL)) {
@@ -31747,6 +32271,10 @@ class HorizontalReduction {
if (ReduxWidth > 1)
ReduxWidth = GetVectorFactor(NumReducedVals - Pos);
AnyVectorized = true;
+ // All remaining reduced values are the identity: nothing left to
+ // vectorize for the accumulator form.
+ if (UseLoopAccForm)
+ break;
}
if (OptReusedScalars && !AnyVectorized) {
for (const std::pair<Value *, unsigned> &P : SameValuesCounter) {
@@ -31763,7 +32291,23 @@ class HorizontalReduction {
if (RK == ReductionOrdering::Ordered)
return VectorizedTree;
- if (!VectorValuesAndScales.empty())
+ SmallVector<std::pair<Instruction *, Value *>> LeftoverReductions;
+ LoopAcc.collectLeftovers(VectorizedVals, ReducedVals, ReducedValsToOps,
+ LeftoverReductions);
+
+ // The accumulator form requires a single vectorized slice and no
+ // leftover reduced values; the latter would have to be folded into the
+ // scalar accumulator on every iteration. Otherwise the accumulator phi is
+ // folded back on top of the reduction like a leftover.
+ Value *AccV = LoopAcc.tryEmit(
+ Builder, RdxFMF, LoopAccVectorized, VectorizedTree, LeftoverReductions,
+ VectorValuesAndScales, RequiredExtract, ReducedValsToOps, ReductionOps,
+ [this, TTI](Value *Vec, IRBuilderBase &B, Type *Ty) {
+ return emitReduction(Vec, B, TTI, Ty);
+ });
+ if (AccV)
+ VectorizedTree = AccV;
+ else if (!VectorValuesAndScales.empty())
VectorizedTree = GetNewVectorizedTree(
VectorizedTree,
emitReduction(Builder, *TTI, ReductionRoot->getType()));
@@ -31877,17 +32421,7 @@ class HorizontalReduction {
SmallVector<std::pair<Instruction *, Value *>> ExtraReductions;
ExtraReductions.emplace_back(cast<Instruction>(ReductionRoot),
VectorizedTree);
- SmallPtrSet<Value *, 8> Visited;
- for (ArrayRef<Value *> Candidates : ReducedVals) {
- for (Value *RdxVal : Candidates) {
- if (!Visited.insert(RdxVal).second)
- continue;
- unsigned NumOps = VectorizedVals.lookup(RdxVal);
- for (Instruction *RedOp :
- ArrayRef(ReducedValsToOps.at(RdxVal)).drop_back(NumOps))
- ExtraReductions.emplace_back(RedOp, RdxVal);
- }
- }
+ ExtraReductions.append(LeftoverReductions);
// Iterate through all not-vectorized reduction values/extra arguments.
bool InitStep = true;
while (ExtraReductions.size() > 1) {
@@ -31898,7 +32432,9 @@ class HorizontalReduction {
}
VectorizedTree = ExtraReductions.front().second;
- ReductionRoot->replaceAllUsesWith(VectorizedTree);
+ // The accumulator emission already replaced all uses of the root.
+ if (!AccV)
+ ReductionRoot->replaceAllUsesWith(VectorizedTree);
// The original scalar reduction is expected to have no remaining
// uses outside the reduction tree itself. Assert that we got this
@@ -32053,9 +32589,11 @@ class HorizontalReduction {
V.calculateTreeCostAndTrimNonProfitable(VL, RdxRootInst);
V.buildExternalUses(LocalExternallyUsedValues);
+ InstructionCost VectorCost, RdxOpCost;
InstructionCost ReductionCost =
getReductionCost(TTI, VL, EmptySameValuesCounter,
- /*IsCmpSelMinMax=*/false, RdxFMF, V, DT, DL, TLI);
+ /*IsCmpSelMinMax=*/false, RdxFMF, V, DT, DL, TLI,
+ VectorCost, RdxOpCost);
InstructionCost Cost =
V.getTreeCost(TreeCost, VL, ReductionCost, RdxRootInst);
LLVM_DEBUG(dbgs() << "SLP: Found cost = " << Cost
@@ -32271,16 +32809,21 @@ class HorizontalReduction {
}
/// Calculate the cost of a reduction.
+ /// \p VectorCost receives the cost of the emitted vector code alone and
+ /// \p RdxOpCost the cost of the reduction operation in it.
InstructionCost getReductionCost(
TargetTransformInfo *TTI, ArrayRef<Value *> ReducedVals,
const SmallMapVector<Value *, unsigned, 16> SameValuesCounter,
bool IsCmpSelMinMax, FastMathFlags FMF, const BoUpSLP &R,
- DominatorTree &DT, const DataLayout &DL, const TargetLibraryInfo &TLI) {
+ DominatorTree &DT, const DataLayout &DL, const TargetLibraryInfo &TLI,
+ InstructionCost &VectorCost, InstructionCost &RdxOpCost) {
const TTI::TargetCostKind CostKind = R.getCostKind();
Type *ScalarTy = ReducedVals.front()->getType();
unsigned ReduxWidth = ReducedVals.size();
FixedVectorType *VectorTy = R.getReductionType();
- InstructionCost VectorCost = 0, ScalarCost;
+ InstructionCost ScalarCost;
+ VectorCost = 0;
+ RdxOpCost = 0;
// If all of the reduced values are constant, the vector cost is 0, since
// the reduction value can be calculated at the compile time.
bool AllConsts = allConstant(ReducedVals);
@@ -32427,6 +32970,7 @@ class HorizontalReduction {
cast<VectorType>(getWidenedType(RType, ReduxWidth)), FMF,
CostKind);
}
+ RdxOpCost = VectorCost;
}
} else {
Type *RedTy = VectorTy->getElementType();
@@ -32504,6 +33048,7 @@ class HorizontalReduction {
if (!AllConsts) {
if (DoesRequireReductionOp) {
VectorCost = TTI->getMinMaxReductionCost(Id, VectorTy, FMF, CostKind);
+ RdxOpCost = VectorCost;
} else {
// Check if the previous reduction already exists and account it as
// series of operations + single reduction.
@@ -32521,6 +33066,7 @@ class HorizontalReduction {
VectorCost += TTI->getCastInstrCost(
Opcode, VectorTy, RVecTy, TTI::CastContextHint::None, CostKind);
}
+ RdxOpCost = VectorCost;
}
}
ScalarCost = EvaluateScalarCost([&](Instruction *RdxOp) {
@@ -33170,7 +33716,7 @@ bool SLPVectorizerPass::vectorizeHorReduction(
HorizontalReduction HorRdx;
Value *Res = nullptr;
if (HorRdx.matchAssociativeReduction(R, Inst, *SE, *DT, *DL, *TTI, *TLI))
- if (Value *Red = HorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT)) {
+ if (Value *Red = HorRdx.tryToReduce(R, *DL, TTI, *TLI, AC, *DT, *LI)) {
if (Red != Inst)
return Red;
Res = Red;
@@ -33339,7 +33885,7 @@ bool SLPVectorizerPass::tryToVectorize(
if (RedCost >= ScalarCost)
return false;
- return HorRdx.tryToReduce(R, *DL, &TTI, *TLI, AC, *DT) != nullptr;
+ return HorRdx.tryToReduce(R, *DL, &TTI, *TLI, AC, *DT, *LI) != nullptr;
};
if (Candidates.size() == 1)
return TryToReduce(I, {Op0, Op1}) ||
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/gather-root.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/gather-root.ll
index 7ae336e2ccee9..681571a54f052 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/gather-root.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/gather-root.ll
@@ -15,10 +15,9 @@ define void @PR28330(i32 %n) {
; DEFAULT-NEXT: [[TMP1:%.*]] = icmp eq <8 x i8> [[TMP0]], zeroinitializer
; DEFAULT-NEXT: br label [[FOR_BODY:%.*]]
; DEFAULT: for.body:
-; DEFAULT-NEXT: [[P17:%.*]] = phi i32 [ [[OP_RDX:%.*]], [[FOR_BODY]] ], [ 0, [[ENTRY:%.*]] ]
+; DEFAULT-NEXT: [[SLPRDX_ACC:%.*]] = phi <8 x i32> [ zeroinitializer, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_BODY]] ]
; DEFAULT-NEXT: [[TMP2:%.*]] = select <8 x i1> [[TMP1]], <8 x i32> splat (i32 -720), <8 x i32> splat (i32 -80)
-; DEFAULT-NEXT: [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP2]])
-; DEFAULT-NEXT: [[OP_RDX]] = add i32 [[TMP3]], [[P17]]
+; DEFAULT-NEXT: [[SLPRDX_ACC1]] = add <8 x i32> [[SLPRDX_ACC]], [[TMP2]]
; DEFAULT-NEXT: br label [[FOR_BODY]]
;
; GATHER-LABEL: @PR28330(
@@ -27,10 +26,9 @@ define void @PR28330(i32 %n) {
; GATHER-NEXT: [[TMP1:%.*]] = icmp eq <8 x i8> [[TMP0]], zeroinitializer
; GATHER-NEXT: br label [[FOR_BODY:%.*]]
; GATHER: for.body:
-; GATHER-NEXT: [[P17:%.*]] = phi i32 [ [[OP_RDX:%.*]], [[FOR_BODY]] ], [ 0, [[ENTRY:%.*]] ]
+; GATHER-NEXT: [[SLPRDX_ACC:%.*]] = phi <8 x i32> [ zeroinitializer, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_BODY]] ]
; GATHER-NEXT: [[TMP2:%.*]] = select <8 x i1> [[TMP1]], <8 x i32> splat (i32 -720), <8 x i32> splat (i32 -80)
-; GATHER-NEXT: [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP2]])
-; GATHER-NEXT: [[OP_RDX]] = add i32 [[TMP3]], [[P17]]
+; GATHER-NEXT: [[SLPRDX_ACC1]] = add <8 x i32> [[SLPRDX_ACC]], [[TMP2]]
; GATHER-NEXT: br label [[FOR_BODY]]
;
; MAX-COST-LABEL: @PR28330(
@@ -39,10 +37,9 @@ define void @PR28330(i32 %n) {
; MAX-COST-NEXT: [[TMP1:%.*]] = icmp eq <8 x i8> [[TMP0]], zeroinitializer
; MAX-COST-NEXT: br label [[FOR_BODY:%.*]]
; MAX-COST: for.body:
-; MAX-COST-NEXT: [[P17:%.*]] = phi i32 [ [[OP_RDX:%.*]], [[FOR_BODY]] ], [ 0, [[ENTRY:%.*]] ]
+; MAX-COST-NEXT: [[SLPRDX_ACC:%.*]] = phi <8 x i32> [ zeroinitializer, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_BODY]] ]
; MAX-COST-NEXT: [[TMP2:%.*]] = select <8 x i1> [[TMP1]], <8 x i32> splat (i32 -720), <8 x i32> splat (i32 -80)
-; MAX-COST-NEXT: [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v8i32(<8 x i32> [[TMP2]])
-; MAX-COST-NEXT: [[OP_RDX]] = add i32 [[TMP3]], [[P17]]
+; MAX-COST-NEXT: [[SLPRDX_ACC1]] = add <8 x i32> [[SLPRDX_ACC]], [[TMP2]]
; MAX-COST-NEXT: br label [[FOR_BODY]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
index 6df79461fb914..d7f5ad7861034 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/getelementptr.ll
@@ -53,12 +53,16 @@ define i32 @getelementptr_4x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y
; CHECK: for.cond.cleanup.loopexit:
; CHECK-NEXT: br label [[FOR_COND_CLEANUP]]
; CHECK: for.cond.cleanup:
-; CHECK-NEXT: [[SUM_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[ADD16:%.*]], [[FOR_COND_CLEANUP_LOOPEXIT:%.*]] ]
-; CHECK-NEXT: ret i32 [[SUM_0_LCSSA]]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x i32> [ poison, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_COND_CLEANUP_LOOPEXIT:%.*]] ]
+; CHECK-NEXT: [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ false, [[ENTRY]] ], [ true, [[FOR_COND_CLEANUP_LOOPEXIT]] ]
+; CHECK-NEXT: [[SUM_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ poison, [[FOR_COND_CLEANUP_LOOPEXIT]] ]
+; CHECK-NEXT: [[TMP6:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[SLPRDX_EXIT]])
+; CHECK-NEXT: [[SLPRDX_SEL:%.*]] = select i1 [[SLPRDX_FROMLOOP]], i32 [[TMP6]], i32 [[SUM_0_LCSSA]]
+; CHECK-NEXT: ret i32 [[SLPRDX_SEL]]
; CHECK: for.body:
-; CHECK-NEXT: [[TMP15:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT: [[SUM_032:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[ADD16]], [[FOR_BODY]] ]
-; CHECK-NEXT: [[T4:%.*]] = shl nsw i32 [[TMP15]], 1
+; CHECK-NEXT: [[SUM_32:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[OP_RDX:%.*]], [[FOR_BODY]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x i32> [ zeroinitializer, [[FOR_BODY_PREHEADER]] ], [ [[SLPRDX_ACC1]], [[FOR_BODY]] ]
+; CHECK-NEXT: [[T4:%.*]] = shl nsw i32 [[SUM_32]], 1
; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x i32> poison, i32 [[T4]], i64 0
; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> zeroinitializer
; CHECK-NEXT: [[TMP3:%.*]] = add nsw <2 x i32> [[TMP2]], [[TMP0]]
@@ -79,10 +83,9 @@ define i32 @getelementptr_4x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y
; CHECK-NEXT: [[TMP18:%.*]] = insertelement <4 x i32> [[TMP17]], i32 [[T8]], i64 1
; CHECK-NEXT: [[TMP19:%.*]] = insertelement <4 x i32> [[TMP18]], i32 [[T10]], i64 2
; CHECK-NEXT: [[TMP9:%.*]] = insertelement <4 x i32> [[TMP19]], i32 [[T12]], i64 3
-; CHECK-NEXT: [[TMP10:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP9]])
-; CHECK-NEXT: [[ADD16]] = add i32 [[TMP10]], [[SUM_032]]
-; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i32 [[TMP15]], 1
-; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i32 [[INDVARS_IV_NEXT]], [[N]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = add <4 x i32> [[SLPRDX_ACC]], [[TMP9]]
+; CHECK-NEXT: [[OP_RDX]] = add nuw nsw i32 [[SUM_32]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i32 [[OP_RDX]], [[N]]
; CHECK-NEXT: br i1 [[EXITCOND]], label [[FOR_COND_CLEANUP_LOOPEXIT]], label [[FOR_BODY]]
;
entry:
@@ -146,12 +149,16 @@ define i32 @getelementptr_2x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y
; CHECK: for.cond.cleanup.loopexit:
; CHECK-NEXT: br label [[FOR_COND_CLEANUP]]
; CHECK: for.cond.cleanup:
-; CHECK-NEXT: [[SUM_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[OP_RDX:%.*]], [[FOR_COND_CLEANUP_LOOPEXIT:%.*]] ]
-; CHECK-NEXT: ret i32 [[SUM_0_LCSSA]]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x i32> [ poison, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_COND_CLEANUP_LOOPEXIT:%.*]] ]
+; CHECK-NEXT: [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ false, [[ENTRY]] ], [ true, [[FOR_COND_CLEANUP_LOOPEXIT]] ]
+; CHECK-NEXT: [[SUM_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ poison, [[FOR_COND_CLEANUP_LOOPEXIT]] ]
+; CHECK-NEXT: [[TMP4:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[SLPRDX_EXIT]])
+; CHECK-NEXT: [[SLPRDX_SEL:%.*]] = select i1 [[SLPRDX_FROMLOOP]], i32 [[TMP4]], i32 [[SUM_0_LCSSA]]
+; CHECK-NEXT: ret i32 [[SLPRDX_SEL]]
; CHECK: for.body:
-; CHECK-NEXT: [[TMP12:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[INDVARS_IV_NEXT:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT: [[SUM_032:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[OP_RDX]], [[FOR_BODY]] ]
-; CHECK-NEXT: [[T4:%.*]] = shl nsw i32 [[TMP12]], 1
+; CHECK-NEXT: [[SUM_32:%.*]] = phi i32 [ 0, [[FOR_BODY_PREHEADER]] ], [ [[OP_RDX1:%.*]], [[FOR_BODY]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x i32> [ zeroinitializer, [[FOR_BODY_PREHEADER]] ], [ [[SLPRDX_ACC1]], [[FOR_BODY]] ]
+; CHECK-NEXT: [[T4:%.*]] = shl nsw i32 [[SUM_32]], 1
; CHECK-NEXT: [[TMP1:%.*]] = insertelement <2 x i32> poison, i32 [[T4]], i64 0
; CHECK-NEXT: [[TMP2:%.*]] = shufflevector <2 x i32> [[TMP1]], <2 x i32> poison, <2 x i32> zeroinitializer
; CHECK-NEXT: [[TMP3:%.*]] = add nsw <2 x i32> [[TMP2]], [[TMP0]]
@@ -168,10 +175,9 @@ define i32 @getelementptr_2x32(ptr nocapture readonly %g, i32 %n, i32 %x, i32 %y
; CHECK-NEXT: [[TMP8:%.*]] = insertelement <4 x i32> [[TMP7]], i32 [[T12]], i64 3
; CHECK-NEXT: [[TMP13:%.*]] = shufflevector <2 x i32> [[TMP5]], <2 x i32> poison, <4 x i32> <i32 0, i32 1, i32 poison, i32 poison>
; CHECK-NEXT: [[TMP14:%.*]] = shufflevector <4 x i32> [[TMP8]], <4 x i32> [[TMP13]], <4 x i32> <i32 4, i32 5, i32 2, i32 3>
-; CHECK-NEXT: [[TMP11:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP14]])
-; CHECK-NEXT: [[OP_RDX]] = add i32 [[TMP11]], [[SUM_032]]
-; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i32 [[TMP12]], 1
-; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i32 [[INDVARS_IV_NEXT]], [[N]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = add <4 x i32> [[SLPRDX_ACC]], [[TMP14]]
+; CHECK-NEXT: [[OP_RDX1]] = add nuw nsw i32 [[SUM_32]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i32 [[OP_RDX1]], [[N]]
; CHECK-NEXT: br i1 [[EXITCOND]], label [[FOR_COND_CLEANUP_LOOPEXIT]], label [[FOR_BODY]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/horizontal.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/horizontal.ll
index 80168812f5fe6..d6ec397aa6c2c 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/horizontal.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/horizontal.ll
@@ -28,8 +28,8 @@ define i32 @test_select(ptr noalias nocapture readonly %blk1, ptr noalias nocapt
; CHECK-NEXT: [[IDX_EXT:%.*]] = sext i32 [[LX:%.*]] to i64
; CHECK-NEXT: br label [[FOR_BODY:%.*]]
; CHECK: for.body:
-; CHECK-NEXT: [[S_026:%.*]] = phi i32 [ 0, [[FOR_BODY_LR_PH]] ], [ [[OP_RDX:%.*]], [[FOR_BODY]] ]
-; CHECK-NEXT: [[J_025:%.*]] = phi i32 [ 0, [[FOR_BODY_LR_PH]] ], [ [[INC:%.*]], [[FOR_BODY]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x i32> [ zeroinitializer, [[FOR_BODY_LR_PH]] ], [ [[SLPRDX_ACC1:%.*]], [[FOR_BODY]] ]
+; CHECK-NEXT: [[J_25:%.*]] = phi i32 [ 0, [[FOR_BODY_LR_PH]] ], [ [[INC1:%.*]], [[FOR_BODY]] ]
; CHECK-NEXT: [[P2_024:%.*]] = phi ptr [ [[BLK2:%.*]], [[FOR_BODY_LR_PH]] ], [ [[ADD_PTR29:%.*]], [[FOR_BODY]] ]
; CHECK-NEXT: [[P1_023:%.*]] = phi ptr [ [[BLK1:%.*]], [[FOR_BODY_LR_PH]] ], [ [[ADD_PTR:%.*]], [[FOR_BODY]] ]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i32>, ptr [[P1_023]], align 4
@@ -38,18 +38,21 @@ define i32 @test_select(ptr noalias nocapture readonly %blk1, ptr noalias nocapt
; CHECK-NEXT: [[TMP3:%.*]] = icmp slt <4 x i32> [[TMP2]], zeroinitializer
; CHECK-NEXT: [[TMP4:%.*]] = sub nsw <4 x i32> zeroinitializer, [[TMP2]]
; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP3]], <4 x i32> [[TMP4]], <4 x i32> [[TMP2]]
-; CHECK-NEXT: [[TMP6:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP5]])
-; CHECK-NEXT: [[OP_RDX]] = add i32 [[TMP6]], [[S_026]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = add <4 x i32> [[SLPRDX_ACC]], [[TMP5]]
; CHECK-NEXT: [[ADD_PTR]] = getelementptr inbounds i32, ptr [[P1_023]], i64 [[IDX_EXT]]
; CHECK-NEXT: [[ADD_PTR29]] = getelementptr inbounds i32, ptr [[P2_024]], i64 [[IDX_EXT]]
-; CHECK-NEXT: [[INC]] = add nuw nsw i32 [[J_025]], 1
-; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i32 [[INC]], [[H]]
+; CHECK-NEXT: [[INC1]] = add nuw nsw i32 [[J_25]], 1
+; CHECK-NEXT: [[EXITCOND:%.*]] = icmp eq i32 [[INC1]], [[H]]
; CHECK-NEXT: br i1 [[EXITCOND]], label [[FOR_END_LOOPEXIT:%.*]], label [[FOR_BODY]]
; CHECK: for.end.loopexit:
; CHECK-NEXT: br label [[FOR_END]]
; CHECK: for.end:
-; CHECK-NEXT: [[S_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY:%.*]] ], [ [[OP_RDX]], [[FOR_END_LOOPEXIT]] ]
-; CHECK-NEXT: ret i32 [[S_0_LCSSA]]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x i32> [ poison, [[ENTRY:%.*]] ], [ [[SLPRDX_ACC1]], [[FOR_END_LOOPEXIT]] ]
+; CHECK-NEXT: [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ false, [[ENTRY]] ], [ true, [[FOR_END_LOOPEXIT]] ]
+; CHECK-NEXT: [[S_0_LCSSA:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ poison, [[FOR_END_LOOPEXIT]] ]
+; CHECK-NEXT: [[TMP6:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[SLPRDX_EXIT]])
+; CHECK-NEXT: [[SLPRDX_SEL:%.*]] = select i1 [[SLPRDX_FROMLOOP]], i32 [[TMP6]], i32 [[S_0_LCSSA]]
+; CHECK-NEXT: ret i32 [[SLPRDX_SEL]]
;
entry:
%cmp.22 = icmp sgt i32 %h, 0
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/loop-accumulator-reduction.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/loop-accumulator-reduction.ll
index 256d16254ea29..3ab8ef2410e2d 100644
--- a/llvm/test/Transforms/SLPVectorizer/AArch64/loop-accumulator-reduction.ll
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/loop-accumulator-reduction.ll
@@ -12,7 +12,7 @@ define double @loop_acc_fadd(ptr %p, i64 %n) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[TMP24:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[IV3:%.*]] = mul nuw i64 [[IV]], 3
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV3]]
; CHECK-NEXT: [[P1:%.*]] = getelementptr double, ptr [[P0]], i64 1
@@ -35,30 +35,30 @@ define double @loop_acc_fadd(ptr %p, i64 %n) {
; CHECK-NEXT: [[TMP3:%.*]] = insertelement <4 x double> [[TMP2]], double [[C3]], i64 1
; CHECK-NEXT: [[TMP4:%.*]] = insertelement <4 x double> [[TMP3]], double [[C6]], i64 2
; CHECK-NEXT: [[TMP5:%.*]] = fmul fast <4 x double> [[TMP1]], [[TMP4]]
-; CHECK-NEXT: [[TMP11:%.*]] = insertelement <4 x double> poison, double [[L1]], i64 0
-; CHECK-NEXT: [[TMP6:%.*]] = insertelement <4 x double> [[TMP11]], double [[ACC]], i64 1
+; CHECK-NEXT: [[TMP6:%.*]] = insertelement <4 x double> <double poison, double 0.000000e+00, double poison, double poison>, double [[L1]], i64 0
; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <4 x double> [[TMP6]], <4 x double> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
; CHECK-NEXT: [[TMP8:%.*]] = insertelement <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, double [[C1]], i64 0
; CHECK-NEXT: [[TMP9:%.*]] = insertelement <4 x double> [[TMP8]], double [[C4]], i64 1
; CHECK-NEXT: [[TMP10:%.*]] = insertelement <4 x double> [[TMP9]], double [[C7]], i64 2
-; CHECK-NEXT: [[TMP12:%.*]] = fmul reassoc nsz arcp contract afn <4 x double> [[TMP7]], [[TMP10]]
-; CHECK-NEXT: [[TMP25:%.*]] = fadd reassoc nsz arcp contract afn <4 x double> [[TMP12]], [[TMP5]]
+; CHECK-NEXT: [[TMP11:%.*]] = fmul fast <4 x double> [[TMP7]], [[TMP10]]
+; CHECK-NEXT: [[TMP12:%.*]] = fadd fast <4 x double> [[TMP11]], [[TMP5]]
; CHECK-NEXT: [[TMP13:%.*]] = insertelement <4 x double> <double poison, double -0.000000e+00, double poison, double poison>, double [[L2]], i64 0
; CHECK-NEXT: [[TMP14:%.*]] = shufflevector <4 x double> [[TMP13]], <4 x double> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
; CHECK-NEXT: [[TMP15:%.*]] = insertelement <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, double [[C2]], i64 0
; CHECK-NEXT: [[TMP16:%.*]] = insertelement <4 x double> [[TMP15]], double [[C5]], i64 1
; CHECK-NEXT: [[TMP17:%.*]] = insertelement <4 x double> [[TMP16]], double [[C8]], i64 2
; CHECK-NEXT: [[TMP18:%.*]] = fmul fast <4 x double> [[TMP14]], [[TMP17]]
-; CHECK-NEXT: [[TMP19:%.*]] = fadd reassoc nsz arcp contract afn <4 x double> [[TMP25]], [[TMP18]]
+; CHECK-NEXT: [[TMP19:%.*]] = fadd fast <4 x double> [[TMP12]], [[TMP18]]
; CHECK-NEXT: [[TMP20:%.*]] = shufflevector <4 x double> [[TMP19]], <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, <4 x i32> <i32 0, i32 1, i32 poison, i32 7>
; CHECK-NEXT: [[TMP21:%.*]] = shufflevector <4 x double> [[TMP20]], <4 x double> [[TMP19]], <4 x i32> <i32 0, i32 1, i32 6, i32 3>
; CHECK-NEXT: [[TMP22:%.*]] = fmul <4 x double> [[TMP19]], [[TMP21]]
-; CHECK-NEXT: [[TMP24]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP22]])
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP22]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP23:%.*]] = phi double [ [[TMP24]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP23:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
; CHECK-NEXT: ret double [[TMP23]]
;
entry:
@@ -511,7 +511,7 @@ define double @two_exit_phis(ptr %p, i64 %n, i1 %c) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[TMP25:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[IV3:%.*]] = mul nuw i64 [[IV]], 3
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV3]]
; CHECK-NEXT: [[P1:%.*]] = getelementptr double, ptr [[P0]], i64 1
@@ -534,31 +534,31 @@ define double @two_exit_phis(ptr %p, i64 %n, i1 %c) {
; CHECK-NEXT: [[TMP3:%.*]] = insertelement <4 x double> [[TMP2]], double [[C3]], i64 1
; CHECK-NEXT: [[TMP4:%.*]] = insertelement <4 x double> [[TMP3]], double [[C6]], i64 2
; CHECK-NEXT: [[TMP5:%.*]] = fmul fast <4 x double> [[TMP1]], [[TMP4]]
-; CHECK-NEXT: [[TMP11:%.*]] = insertelement <4 x double> poison, double [[L1]], i64 0
-; CHECK-NEXT: [[TMP6:%.*]] = insertelement <4 x double> [[TMP11]], double [[ACC]], i64 1
+; CHECK-NEXT: [[TMP6:%.*]] = insertelement <4 x double> <double poison, double 0.000000e+00, double poison, double poison>, double [[L1]], i64 0
; CHECK-NEXT: [[TMP7:%.*]] = shufflevector <4 x double> [[TMP6]], <4 x double> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
; CHECK-NEXT: [[TMP8:%.*]] = insertelement <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, double [[C1]], i64 0
; CHECK-NEXT: [[TMP9:%.*]] = insertelement <4 x double> [[TMP8]], double [[C4]], i64 1
; CHECK-NEXT: [[TMP10:%.*]] = insertelement <4 x double> [[TMP9]], double [[C7]], i64 2
-; CHECK-NEXT: [[TMP12:%.*]] = fmul reassoc nsz arcp contract afn <4 x double> [[TMP7]], [[TMP10]]
-; CHECK-NEXT: [[TMP26:%.*]] = fadd reassoc nsz arcp contract afn <4 x double> [[TMP12]], [[TMP5]]
+; CHECK-NEXT: [[TMP11:%.*]] = fmul fast <4 x double> [[TMP7]], [[TMP10]]
+; CHECK-NEXT: [[TMP12:%.*]] = fadd fast <4 x double> [[TMP11]], [[TMP5]]
; CHECK-NEXT: [[TMP13:%.*]] = insertelement <4 x double> <double poison, double -0.000000e+00, double poison, double poison>, double [[L2]], i64 0
; CHECK-NEXT: [[TMP14:%.*]] = shufflevector <4 x double> [[TMP13]], <4 x double> poison, <4 x i32> <i32 0, i32 0, i32 0, i32 1>
; CHECK-NEXT: [[TMP15:%.*]] = insertelement <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, double [[C2]], i64 0
; CHECK-NEXT: [[TMP16:%.*]] = insertelement <4 x double> [[TMP15]], double [[C5]], i64 1
; CHECK-NEXT: [[TMP17:%.*]] = insertelement <4 x double> [[TMP16]], double [[C8]], i64 2
; CHECK-NEXT: [[TMP18:%.*]] = fmul fast <4 x double> [[TMP14]], [[TMP17]]
-; CHECK-NEXT: [[TMP19:%.*]] = fadd reassoc nsz arcp contract afn <4 x double> [[TMP26]], [[TMP18]]
+; CHECK-NEXT: [[TMP19:%.*]] = fadd fast <4 x double> [[TMP12]], [[TMP18]]
; CHECK-NEXT: [[TMP20:%.*]] = shufflevector <4 x double> [[TMP19]], <4 x double> <double poison, double poison, double poison, double 1.000000e+00>, <4 x i32> <i32 0, i32 1, i32 poison, i32 7>
; CHECK-NEXT: [[TMP21:%.*]] = shufflevector <4 x double> [[TMP20]], <4 x double> [[TMP19]], <4 x i32> <i32 0, i32 1, i32 6, i32 3>
; CHECK-NEXT: [[TMP22:%.*]] = fmul <4 x double> [[TMP19]], [[TMP21]]
-; CHECK-NEXT: [[TMP25]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP22]])
+; CHECK-NEXT: [[TMP25:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP22]])
+; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP25]], [[ACC]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP23:%.*]] = phi double [ [[TMP25]], %[[LOOP]] ]
-; CHECK-NEXT: [[TMP24:%.*]] = phi double [ [[TMP25]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP23:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP24:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
; CHECK-NEXT: [[R:%.*]] = fadd double [[TMP23]], [[TMP24]]
; CHECK-NEXT: ret double [[R]]
;
@@ -622,21 +622,22 @@ define double @dup_preheader_edges(ptr %p, i64 %n, double %init, i32 %sw) {
; CHECK-LABEL: define double @dup_preheader_edges(
; CHECK-SAME: ptr [[P:%.*]], i64 [[N:%.*]], double [[INIT:%.*]], i32 [[SW:%.*]]) {
; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[SLPRDX_INIT:%.*]] = insertelement <4 x double> zeroinitializer, double [[INIT]], i32 0
; CHECK-NEXT: switch i32 [[SW]], label %[[LOOP:.*]] [
; CHECK-NEXT: i32 1, label %[[LOOP]]
; CHECK-NEXT: ]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ [[INIT]], %[[ENTRY]] ], [ [[INIT]], %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ [[SLPRDX_INIT]], %[[ENTRY]] ], [ [[SLPRDX_INIT]], %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
; CHECK-NEXT: ret double [[TMP1]]
;
entry:
@@ -679,16 +680,19 @@ define double @dup_exit_edges_bypass(ptr %p, i64 %n, double %y, i32 %sw) {
; CHECK-NEXT: ]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ], [ poison, %[[ENTRY]] ], [ poison, %[[ENTRY]] ]
+; CHECK-NEXT: [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LOOP]] ], [ false, %[[ENTRY]] ], [ false, %[[ENTRY]] ]
+; CHECK-NEXT: [[RES:%.*]] = phi double [ poison, %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT: [[TMP1:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP2]], double [[RES]]
; CHECK-NEXT: ret double [[TMP1]]
;
entry:
@@ -777,11 +781,10 @@ define double @early_exit_in_loop_value(ptr %p, i64 %n) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LATCH]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LATCH]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP3]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x double> [[TMP0]], i64 0
; CHECK-NEXT: [[EC:%.*]] = fcmp ogt double [[TMP1]], 1.000000e+10
; CHECK-NEXT: br i1 [[EC]], label %[[EXIT:.*]], label %[[LATCH]]
@@ -790,7 +793,11 @@ define double @early_exit_in_loop_value(ptr %p, i64 %n) {
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP2:%.*]] = phi double [ [[OP_RDX]], %[[LATCH]] ], [ [[TMP1]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LATCH]] ], [ poison, %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LATCH]] ], [ false, %[[LOOP]] ]
+; CHECK-NEXT: [[RES:%.*]] = phi double [ poison, %[[LATCH]] ], [ [[TMP1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT: [[TMP2:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP3]], double [[RES]]
; CHECK-NEXT: ret double [[TMP2]]
;
entry:
@@ -1295,11 +1302,10 @@ define double @bypass_from_sibling_loop(ptr %p, ptr %q, i64 %n, i64 %m, i1 %c) {
; CHECK-NEXT: br i1 [[C]], label %[[LOOP:.*]], label %[[LOOP2:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP1]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
@@ -1313,7 +1319,11 @@ define double @bypass_from_sibling_loop(ptr %p, ptr %q, i64 %n, i64 %m, i1 %c) {
; CHECK-NEXT: [[CMP2:%.*]] = icmp eq i64 [[J_NEXT]], [[M]]
; CHECK-NEXT: br i1 [[CMP2]], label %[[EXIT]], label %[[LOOP2]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[RES:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ], [ [[SUM2]], %[[LOOP2]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ], [ poison, %[[LOOP2]] ]
+; CHECK-NEXT: [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LOOP]] ], [ false, %[[LOOP2]] ]
+; CHECK-NEXT: [[RES1:%.*]] = phi double [ poison, %[[LOOP]] ], [ [[SUM2]], %[[LOOP2]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT: [[RES:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP1]], double [[RES1]]
; CHECK-NEXT: ret double [[RES]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/loop-accumulator-reduction.ll b/llvm/test/Transforms/SLPVectorizer/X86/loop-accumulator-reduction.ll
index f05cc887d7315..55a31632227eb 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/loop-accumulator-reduction.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/loop-accumulator-reduction.ll
@@ -16,16 +16,16 @@ define double @loop_acc_fadd(ptr %p) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
; CHECK-NEXT: ret double [[TMP1]]
;
entry:
@@ -217,17 +217,18 @@ define double @two_exit_phis(ptr %p) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP3]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
-; CHECK-NEXT: [[TMP2:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT2:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT2]])
+; CHECK-NEXT: [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
; CHECK-NEXT: [[R:%.*]] = fadd double [[TMP1]], [[TMP2]]
; CHECK-NEXT: ret double [[R]]
;
@@ -270,16 +271,16 @@ define double @loop_acc_fadd_unknown_trip_count(ptr %p, i64 %n) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], [[N]]
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
; CHECK-NEXT: ret double [[TMP1]]
;
entry:
@@ -319,16 +320,16 @@ define double @loop_acc_fmax(ptr %p, i64 %n) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[TMP3:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ <double 0.000000e+00, double -inf, double -inf, double -inf>, %[[ENTRY]] ], [ [[TMP1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = call nnan double @llvm.vector.reduce.fmax.v4f64(<4 x double> [[TMP0]])
-; CHECK-NEXT: [[TMP3]] = call nnan double @llvm.maxnum.f64(double [[TMP1]], double [[ACC]])
+; CHECK-NEXT: [[TMP1]] = call nnan <4 x double> @llvm.maxnum.v4f64(<4 x double> [[SLPRDX_ACC]], <4 x double> [[TMP0]])
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP2:%.*]] = phi double [ [[TMP3]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[TMP1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = call nnan double @llvm.vector.reduce.fmax.v4f64(<4 x double> [[SLPRDX_EXIT]])
; CHECK-NEXT: ret double [[TMP2]]
;
entry:
@@ -365,21 +366,22 @@ define double @dup_preheader_edges(ptr %p, i64 %n, double %init, i32 %sw) {
; CHECK-LABEL: define double @dup_preheader_edges(
; CHECK-SAME: ptr [[P:%.*]], i64 [[N:%.*]], double [[INIT:%.*]], i32 [[SW:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[SLPRDX_INIT:%.*]] = insertelement <4 x double> zeroinitializer, double [[INIT]], i32 0
; CHECK-NEXT: switch i32 [[SW]], label %[[LOOP:.*]] [
; CHECK-NEXT: i32 1, label %[[LOOP]]
; CHECK-NEXT: ]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ [[INIT]], %[[ENTRY]] ], [ [[INIT]], %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ [[SLPRDX_INIT]], %[[ENTRY]] ], [ [[SLPRDX_INIT]], %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
; CHECK-NEXT: ret double [[TMP1]]
;
entry:
@@ -423,16 +425,19 @@ define double @dup_exit_edges_bypass(ptr %p, i64 %n, double %y, i32 %sw) {
; CHECK-NEXT: ]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP1]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[SLPRDX_SEL:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ], [ poison, %[[ENTRY]] ], [ poison, %[[ENTRY]] ]
+; CHECK-NEXT: [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LOOP]] ], [ false, %[[ENTRY]] ], [ false, %[[ENTRY]] ]
+; CHECK-NEXT: [[RES:%.*]] = phi double [ poison, %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT: [[SLPRDX_SEL:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP1]], double [[RES]]
; CHECK-NEXT: ret double [[SLPRDX_SEL]]
;
entry:
@@ -474,16 +479,19 @@ define double @fmax_bypass(ptr %p, i64 %n, double %y, i1 %c) {
; CHECK-NEXT: br i1 [[C]], label %[[LOOP:.*]], label %[[EXIT:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[TMP2:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ <double 0.000000e+00, double f0xFFEFFFFFFFFFFFFF, double f0xFFEFFFFFFFFFFFFF, double f0xFFEFFFFFFFFFFFFF>, %[[ENTRY]] ], [ [[TMP1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = call nnan ninf double @llvm.vector.reduce.fmax.v4f64(<4 x double> [[TMP0]])
-; CHECK-NEXT: [[TMP2]] = call nnan ninf double @llvm.maxnum.f64(double [[TMP1]], double [[ACC]])
+; CHECK-NEXT: [[TMP1]] = call nnan ninf <4 x double> @llvm.maxnum.v4f64(<4 x double> [[SLPRDX_ACC]], <4 x double> [[TMP0]])
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[SLPRDX_SEL:%.*]] = phi double [ [[TMP2]], %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[TMP1]], %[[LOOP]] ], [ poison, %[[ENTRY]] ]
+; CHECK-NEXT: [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LOOP]] ], [ false, %[[ENTRY]] ]
+; CHECK-NEXT: [[RES:%.*]] = phi double [ poison, %[[LOOP]] ], [ [[Y]], %[[ENTRY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = call nnan ninf double @llvm.vector.reduce.fmax.v4f64(<4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT: [[SLPRDX_SEL:%.*]] = select nnan ninf i1 [[SLPRDX_FROMLOOP]], double [[TMP2]], double [[RES]]
; CHECK-NEXT: ret double [[SLPRDX_SEL]]
;
entry:
@@ -523,11 +531,10 @@ define double @early_exit_in_loop_value(ptr %p, i64 %n) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LATCH]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LATCH]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x double> [[TMP0]], i64 0
; CHECK-NEXT: [[EC:%.*]] = fcmp ogt double [[TMP1]], 1.000000e+10
; CHECK-NEXT: br i1 [[EC]], label %[[EXIT:.*]], label %[[LATCH]]
@@ -536,7 +543,11 @@ define double @early_exit_in_loop_value(ptr %p, i64 %n) {
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[SLPRDX_SEL:%.*]] = phi double [ [[OP_RDX]], %[[LATCH]] ], [ [[TMP1]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LATCH]] ], [ poison, %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LATCH]] ], [ false, %[[LOOP]] ]
+; CHECK-NEXT: [[RES:%.*]] = phi double [ poison, %[[LATCH]] ], [ [[TMP1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT: [[SLPRDX_SEL:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP2]], double [[RES]]
; CHECK-NEXT: ret double [[SLPRDX_SEL]]
;
entry:
@@ -645,14 +656,14 @@ define double @rdx_op_outside_loop(ptr %p, i64 %n) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
; CHECK-NEXT: ret double [[TMP1]]
;
entry:
@@ -1050,11 +1061,10 @@ define double @bypass_from_sibling_loop(ptr %p, ptr %q, i64 %n, i64 %m, i1 %c) {
; CHECK-NEXT: br i1 [[C]], label %[[LOOP:.*]], label %[[LOOP2:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP1]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
@@ -1068,7 +1078,11 @@ define double @bypass_from_sibling_loop(ptr %p, ptr %q, i64 %n, i64 %m, i1 %c) {
; CHECK-NEXT: [[CMP2:%.*]] = icmp eq i64 [[J_NEXT]], [[M]]
; CHECK-NEXT: br i1 [[CMP2]], label %[[EXIT]], label %[[LOOP2]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[SLPRDX_SEL:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ], [ [[SUM2]], %[[LOOP2]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ], [ poison, %[[LOOP2]] ]
+; CHECK-NEXT: [[SLPRDX_FROMLOOP:%.*]] = phi i1 [ true, %[[LOOP]] ], [ false, %[[LOOP2]] ]
+; CHECK-NEXT: [[RES:%.*]] = phi double [ poison, %[[LOOP]] ], [ [[SUM2]], %[[LOOP2]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
+; CHECK-NEXT: [[SLPRDX_SEL:%.*]] = select fast i1 [[SLPRDX_FROMLOOP]], double [[TMP1]], double [[RES]]
; CHECK-NEXT: ret double [[SLPRDX_SEL]]
;
entry:
@@ -1117,19 +1131,19 @@ define double @dot_product(ptr %p, ptr %q) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[Q0:%.*]] = getelementptr double, ptr [[Q]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
; CHECK-NEXT: [[TMP1:%.*]] = load <4 x double>, ptr [[Q0]], align 8
; CHECK-NEXT: [[TMP2:%.*]] = fmul fast <4 x double> [[TMP0]], [[TMP1]]
-; CHECK-NEXT: [[TMP4:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP2]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP4]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP2]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP3:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
; CHECK-NEXT: ret double [[TMP3]]
;
entry:
@@ -1232,17 +1246,17 @@ define double @call_in_loop_fp(ptr %p) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi double [ 0.000000e+00, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr double, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x double>, ptr [[P0]], align 8
-; CHECK-NEXT: [[TMP2:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP0]])
-; CHECK-NEXT: [[OP_RDX]] = fadd fast double [[TMP2]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP0]]
; CHECK-NEXT: call void @sink()
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP1:%.*]] = phi double [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x double> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[SLPRDX_EXIT]])
; CHECK-NEXT: ret double [[TMP1]]
;
entry:
@@ -1282,17 +1296,17 @@ define i32 @zext_leaves(ptr %p) {
; CHECK-NEXT: br label %[[LOOP:.*]]
; CHECK: [[LOOP]]:
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT: [[ACC:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[OP_RDX:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x i32> [ zeroinitializer, %[[ENTRY]] ], [ [[SLPRDX_ACC1:%.*]], %[[LOOP]] ]
; CHECK-NEXT: [[P0:%.*]] = getelementptr i16, ptr [[P]], i64 [[IV]]
; CHECK-NEXT: [[TMP0:%.*]] = load <4 x i16>, ptr [[P0]], align 2
; CHECK-NEXT: [[TMP1:%.*]] = zext <4 x i16> [[TMP0]] to <4 x i32>
-; CHECK-NEXT: [[TMP3:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[TMP1]])
-; CHECK-NEXT: [[OP_RDX]] = add i32 [[TMP3]], [[ACC]]
+; CHECK-NEXT: [[SLPRDX_ACC1]] = add <4 x i32> [[SLPRDX_ACC]], [[TMP1]]
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[CMP:%.*]] = icmp eq i64 [[IV]], 64
; CHECK-NEXT: br i1 [[CMP]], label %[[EXIT:.*]], label %[[LOOP]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[TMP2:%.*]] = phi i32 [ [[OP_RDX]], %[[LOOP]] ]
+; CHECK-NEXT: [[SLPRDX_EXIT:%.*]] = phi <4 x i32> [ [[SLPRDX_ACC1]], %[[LOOP]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> [[SLPRDX_EXIT]])
; CHECK-NEXT: ret i32 [[TMP2]]
;
entry:
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/slp-fma-loss.ll b/llvm/test/Transforms/SLPVectorizer/X86/slp-fma-loss.ll
index f48748dd137b6..1a7b3a9ff170c 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/slp-fma-loss.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/slp-fma-loss.ll
@@ -11,12 +11,11 @@ define void @hr() {
; SSE4-LABEL: @hr(
; SSE4-NEXT: br label [[LOOP:%.*]]
; SSE4: loop:
-; SSE4-NEXT: [[PHI0:%.*]] = phi double [ 0.000000e+00, [[TMP0:%.*]] ], [ [[OP_RDX:%.*]], [[LOOP]] ]
+; SSE4-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, [[TMP0:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[LOOP]] ]
; SSE4-NEXT: [[CVT0:%.*]] = uitofp i16 0 to double
; SSE4-NEXT: [[TMP1:%.*]] = insertelement <4 x double> <double poison, double 0.000000e+00, double 0.000000e+00, double 0.000000e+00>, double [[CVT0]], i64 0
; SSE4-NEXT: [[TMP2:%.*]] = fmul fast <4 x double> zeroinitializer, [[TMP1]]
-; SSE4-NEXT: [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP2]])
-; SSE4-NEXT: [[OP_RDX]] = fadd fast double [[TMP3]], [[PHI0]]
+; SSE4-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP2]]
; SSE4-NEXT: br i1 true, label [[EXIT:%.*]], label [[LOOP]]
; SSE4: exit:
; SSE4-NEXT: ret void
@@ -24,13 +23,11 @@ define void @hr() {
; AVX-LABEL: @hr(
; AVX-NEXT: br label [[LOOP:%.*]]
; AVX: loop:
-; AVX-NEXT: [[PHI0:%.*]] = phi double [ 0.000000e+00, [[TMP0:%.*]] ], [ [[ADD3:%.*]], [[LOOP]] ]
+; AVX-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, [[TMP0:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[LOOP]] ]
; AVX-NEXT: [[CVT0:%.*]] = uitofp i16 0 to double
-; AVX-NEXT: [[MUL0:%.*]] = fmul fast double 0.000000e+00, [[CVT0]]
-; AVX-NEXT: [[ADD0:%.*]] = fadd fast double [[MUL0]], [[PHI0]]
-; AVX-NEXT: [[ADD1:%.*]] = fadd fast double 0.000000e+00, [[ADD0]]
-; AVX-NEXT: [[ADD2:%.*]] = fadd fast double 0.000000e+00, [[ADD1]]
-; AVX-NEXT: [[ADD3]] = fadd fast double 0.000000e+00, [[ADD2]]
+; AVX-NEXT: [[TMP1:%.*]] = insertelement <4 x double> <double poison, double 0.000000e+00, double 0.000000e+00, double 0.000000e+00>, double [[CVT0]], i64 0
+; AVX-NEXT: [[TMP2:%.*]] = fmul fast <4 x double> zeroinitializer, [[TMP1]]
+; AVX-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP2]]
; AVX-NEXT: br i1 true, label [[EXIT:%.*]], label [[LOOP]]
; AVX: exit:
; AVX-NEXT: ret void
@@ -38,12 +35,11 @@ define void @hr() {
; AVX2-LABEL: @hr(
; AVX2-NEXT: br label [[LOOP:%.*]]
; AVX2: loop:
-; AVX2-NEXT: [[PHI0:%.*]] = phi double [ 0.000000e+00, [[TMP0:%.*]] ], [ [[OP_RDX:%.*]], [[LOOP]] ]
+; AVX2-NEXT: [[SLPRDX_ACC:%.*]] = phi <4 x double> [ zeroinitializer, [[TMP0:%.*]] ], [ [[SLPRDX_ACC1:%.*]], [[LOOP]] ]
; AVX2-NEXT: [[CVT0:%.*]] = uitofp i16 0 to double
; AVX2-NEXT: [[TMP1:%.*]] = insertelement <4 x double> <double poison, double 0.000000e+00, double 0.000000e+00, double 0.000000e+00>, double [[CVT0]], i64 0
; AVX2-NEXT: [[TMP2:%.*]] = fmul fast <4 x double> zeroinitializer, [[TMP1]]
-; AVX2-NEXT: [[TMP3:%.*]] = call fast double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> [[TMP2]])
-; AVX2-NEXT: [[OP_RDX]] = fadd fast double [[TMP3]], [[PHI0]]
+; AVX2-NEXT: [[SLPRDX_ACC1]] = fadd reassoc nnan ninf nsz arcp afn <4 x double> [[SLPRDX_ACC]], [[TMP2]]
; AVX2-NEXT: br i1 true, label [[EXIT:%.*]], label [[LOOP]]
; AVX2: exit:
; AVX2-NEXT: ret void
More information about the llvm-commits
mailing list