[llvm] [SLP]Emit loop-carried horizontal reductions as loop vector accumulator (PR #221598)
Ryan Buchner via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 20 09:28:19 PDT 2026
================
@@ -31063,12 +31073,502 @@ class HorizontalReduction {
return true;
}
+ /// Loop accumulator emission mode of the reduction: a reduction that
+ /// accumulates a loop phi (the phi is one of the reduced values and the
+ /// root is its backedge value) can be emitted as a vector accumulator in
+ /// the loop plus a single reduction after the loop instead of a horizontal
+ /// reduction on every iteration.
+ class LoopAccumulator {
+ /// The reduction root and the reduction operation kind and ordering.
+ Instruction *Root = nullptr;
+ RecurKind RdxKind = RecurKind::None;
+ ReductionOrdering RK = ReductionOrdering::None;
+ /// The vectorizer state and the analyses the accumulator cost and
+ /// emission are computed with.
+ BoUpSLP &R;
+ const TargetTransformInfo &TTI;
+ LoopInfo &LI;
+ /// The accumulated phi and the identity constant that replaced it in the
+ /// reduced values. The identity is never folded as a leftover: it does
+ /// not change the result.
+ PHINode *Phi = nullptr;
+ Constant *Identity = nullptr;
+ /// The number of calls in the loop, across which the vector accumulator
+ /// is kept live.
+ unsigned NumCalls = 0;
+ /// Exit phis consuming the reduction result outside the loop.
+ SmallSetVector<PHINode *, 2> ExitPhis;
+
+ /// \returns the identity constant of the reduction operation for \p Ty
+ /// under the fast-math flags \p FMF.
+ Constant *getIdentity(Type *Ty, FastMathFlags FMF) const {
+ Intrinsic::ID Id;
+ switch (RdxKind) {
+ case RecurKind::FMax:
+ case RecurKind::FMaxNum:
+ Id = Intrinsic::vector_reduce_fmax;
+ break;
+ case RecurKind::FMin:
+ case RecurKind::FMinNum:
+ Id = Intrinsic::vector_reduce_fmin;
+ break;
+ case RecurKind::FMaximum:
+ Id = Intrinsic::vector_reduce_fmaximum;
+ break;
+ case RecurKind::FMinimum:
+ Id = Intrinsic::vector_reduce_fminimum;
+ break;
+ default:
+ Id = RecurrenceDescriptor::isIntMinMaxRecurrenceKind(RdxKind)
+ ? getMinMaxReductionIntrinsicID(
+ getMinMaxReductionIntrinsicOp(RdxKind))
+ : getReductionForBinop(static_cast<Instruction::BinaryOps>(
+ RecurrenceDescriptor::getOpcode(RdxKind)));
+ break;
+ }
+ return cast<Constant>(getReductionIdentity(Id, Ty, FMF));
+ }
+
+ /// \returns the extra cost of accumulating the slice lane-wise into the
+ /// vector accumulator instead of reducing it on every iteration: the
+ /// lane-wise operation, the spill and reload on every iteration of the
+ /// values live beyond the register file and keeping the vector
+ /// accumulator live across the calls of the loop.
+ InstructionCost getAccumulationCost(FastMathFlags FMF, VectorType *VectorTy,
+ Type *ScalarTy) const {
+ const TTI::TargetCostKind CostKind = R.getCostKind();
+ auto GetOpCost = [&](Type *OpTy) {
+ if (RecurrenceDescriptor::isMinMaxRecurrenceKind(RdxKind)) {
+ IntrinsicCostAttributes ICA(getMinMaxReductionIntrinsicOp(RdxKind),
+ OpTy, {OpTy, OpTy}, FMF);
+ return TTI.getIntrinsicInstrCost(ICA, CostKind);
+ }
+ return TTI.getArithmeticInstrCost(
+ RecurrenceDescriptor::getOpcode(RdxKind), OpTy, CostKind);
+ };
+ // The lane-wise operation; also replaces the scalar operation folding
+ // the accumulator phi on top of the horizontal reduction (the scalar
+ // cost of the slice credits it neither, the slice lacks the phi).
+ InstructionCost Cost = GetOpCost(VectorTy) - GetOpCost(ScalarTy);
+ // The vector accumulator and the reduced vector are live at the same
+ // time; the registers they need beyond the register file are spilled
+ // and reloaded on every iteration.
+ constexpr unsigned NumLiveVectors = 2;
+ unsigned Regs = TTI.getRegUsageForType(VectorTy);
+ unsigned RC = TTI.getRegisterClassForType(/*Vector=*/true, VectorTy);
+ unsigned NumRegs = TTI.getNumberOfRegisters(RC);
+ InstructionCost SpillReload =
+ TTI.getRegisterClassSpillCost(RC, CostKind) +
+ TTI.getRegisterClassReloadCost(RC, CostKind);
+ InstructionCost Spilled = 0;
+ if (NumRegs != 0 && NumLiveVectors * Regs > NumRegs)
+ Spilled = SpillReload * (NumLiveVectors * Regs - NumRegs);
+ Cost += Spilled;
+ // The vector accumulator replaces the scalar one across the calls of
+ // the loop: without callee-saved registers of the class it is spilled
+ // and reloaded around every call, like the scalar one. The accumulator
+ // registers already spilled on every iteration stay in memory across
+ // the calls and are not spilled again.
+ if (NumCalls != 0) {
+ InstructionCost VecLive =
+ std::max(TTI.getCostOfKeepingLiveOverCall(VectorTy),
+ SpillReload * Regs) -
+ std::min(Spilled, SpillReload * Regs);
+ unsigned SRC = TTI.getRegisterClassForType(/*Vector=*/false, ScalarTy);
+ InstructionCost ScalarLive =
+ std::max(TTI.getCostOfKeepingLiveOverCall(ScalarTy),
+ TTI.getRegisterClassSpillCost(SRC, CostKind) +
+ TTI.getRegisterClassReloadCost(SRC, CostKind));
+ Cost += (VecLive - ScalarLive) * NumCalls;
+ }
+ return Cost;
+ }
+
+ public:
+ LoopAccumulator(Instruction *Root, RecurKind RdxKind, ReductionOrdering RK,
+ BoUpSLP &R, const TargetTransformInfo &TTI, LoopInfo &LI)
+ : Root(Root), RdxKind(RdxKind), RK(RK), R(R), TTI(TTI), LI(LI) {}
+
+ /// If the reduction accumulates a loop phi (the phi is one of the reduced
+ /// values and the root is its backedge value), replaces the phi by the
+ /// reduction identity in the reduced values, so that the emitted tree
+ /// does not reference the phi. The identity does not change the result,
+ /// so if the accumulator emission does not apply, the phi is just folded
+ /// back on top of the emitted reduction. All unordered reduction kinds
+ /// have an identity constant.
+ void prepare(FastMathFlags RdxFMF,
+ SmallVectorImpl<SmallVector<Value *>> &ReducedVals,
+ SmallDenseMap<Value *, SmallVector<Instruction *>, 16>
+ &ReducedValsToOps,
+ const SmallPtrSetImpl<Value *> &NegatedReducedVals,
+ bool HasNarrowedLeafShifts) {
+ if (!VectorizeLoopAccRdx || RK != ReductionOrdering::Unordered ||
+ Root->getType()->isIntegerTy(1) || Root->getType()->isVectorTy() ||
+ HasNarrowedLeafShifts || isCmpSelMinMax(Root) ||
+ Root->hasNUsesOrMore(UsesLimit))
+ return;
+ Loop *L = LI.getLoopFor(Root->getParent());
+ if (!L)
+ return;
+ BasicBlock *Latch = L->getLoopLatch();
+ if (!Latch)
+ return;
+ // Only a single accumulated phi, used by the reduction operations
+ // only, is supported.
+ auto FindAccPhi = [&]() -> PHINode * {
+ PHINode *AccPhi = nullptr;
+ for (ArrayRef<Value *> Candidates : ReducedVals)
+ for (Value *RdxVal : Candidates) {
+ auto *P = dyn_cast<PHINode>(RdxVal);
+ if (!P || P->getParent() != L->getHeader() ||
+ P->getNumIncomingValues() > MaxPHINumOperands ||
+ P->getIncomingValueForBlock(Latch) != Root)
+ continue;
+ if (AccPhi || P->hasNUsesOrMore(UsesLimit) ||
+ P->getNumUses() != ReducedValsToOps.at(P).size())
+ return nullptr;
+ AccPhi = P;
+ }
+ return AccPhi;
+ };
+ PHINode *AccPhi = FindAccPhi();
+ // A negated phi is subtracted from the reduction result, not
+ // accumulated.
+ if (!AccPhi || NegatedReducedVals.contains(AccPhi))
+ return;
+ Constant *IdC = getIdentity(AccPhi->getType(), RdxFMF);
+ if (ReducedValsToOps.contains(IdC))
+ return;
+ // The root may be used only by the accumulator phi and by phis outside
+ // the loop (the exit phis). The initial values are inserted into the
+ // identity vector at the end of their incoming blocks, so they must not
+ // be defined by the terminators; the final reduction is emitted at the
+ // beginning of the exit blocks, which must not be EH pads. Neither may
+ // land in a loop not containing the reduction loop, which would execute
+ // them repeatedly.
+ auto IsExecutedOncePerLoop = [&](BasicBlock *BB) {
+ Loop *BBL = LI.getLoopFor(BB);
+ return !BBL || BBL->contains(L);
+ };
+ if (any_of(zip(AccPhi->incoming_values(), AccPhi->blocks()),
+ [&](const auto &P) {
+ auto [IncV, B] = P;
+ return IncV != Root && (IncV == B->getTerminator() ||
+ !IsExecutedOncePerLoop(B));
+ }))
+ return;
+ auto IsValidExitPhi = [&](User *U) {
+ auto *ExitPhi = dyn_cast<PHINode>(U);
+ return ExitPhi &&
+ ExitPhi->getNumIncomingValues() <= MaxPHINumOperands &&
+ !ExitPhi->getParent()->isEHPad() &&
+ !L->contains(ExitPhi->getParent()) &&
+ IsExecutedOncePerLoop(ExitPhi->getParent());
+ };
+ auto CollectExitPhis = [&] {
+ for (User *U : Root->users()) {
+ if (U == AccPhi)
+ continue;
+ if (!IsValidExitPhi(U))
+ return false;
+ ExitPhis.insert(cast<PHINode>(U));
+ }
+ return true;
+ };
+ if (!CollectExitPhis())
+ return;
+ // The final reduction and the insert of the initial value would be on
+ // the loop-carried chain of an enclosing loop if the reduction result
+ // feeds the initial value of the accumulator through it: keep the
+ // horizontal reduction, whose chain is the scalar accumulator only.
+ auto IsFedByExitPhi = [&](Value *InitV) {
+ auto *InitPhi = dyn_cast<PHINode>(InitV);
+ if (InitV == Root || !InitPhi)
+ return false;
+ // Too many incoming values to scan: keep the horizontal reduction.
+ if (InitPhi->getNumIncomingValues() > MaxPHINumOperands)
+ return true;
+ return ExitPhis.contains(InitPhi) ||
+ any_of(InitPhi->incoming_values(), [&](Value *V) {
+ return isa<PHINode>(V) && ExitPhis.contains(cast<PHINode>(V));
+ });
+ };
+ if (any_of(AccPhi->incoming_values(), IsFedByExitPhi))
+ return;
+ // Calls clobber the vector registers: the vector accumulator has to be
+ // kept live across them. Intrinsics cheaper than a call are not calls.
+ for (BasicBlock *BB : L->blocks())
+ NumCalls += count_if(*BB, [&](const Instruction &I) {
+ auto *CB = dyn_cast<CallBase>(&I);
+ if (!CB || CB->doesNotReturn())
+ return false;
+ auto *II = dyn_cast<IntrinsicInst>(CB);
+ if (!II)
+ return true;
+ if (II->isAssumeLikeIntrinsic())
+ return false;
+ IntrinsicCostAttributes ICA(II->getIntrinsicID(), *II);
+ return TTI.getIntrinsicInstrCost(ICA, R.getCostKind()) >=
+ TTI.getCallInstrCost(nullptr, II->getType(), ICA.getArgTypes(),
+ R.getCostKind());
+ });
+ for (SmallVector<Value *> &Candidates : ReducedVals)
+ for (Value *&RdxVal : Candidates)
+ if (RdxVal == AccPhi)
+ RdxVal = IdC;
+ ReducedValsToOps.try_emplace(IdC, ReducedValsToOps.lookup(AccPhi));
+ ReducedValsToOps.erase(AccPhi);
+ Phi = AccPhi;
+ Identity = IdC;
+ }
+
+ /// Collects the reduced values that were not vectorized; they are folded
+ /// into the reduction on top of the vectorized part. The accumulator
+ /// identity constant is never folded.
+ void collectLeftovers(
----------------
bababuck wrote:
Figured out my confusion.
https://github.com/llvm/llvm-project/pull/221598
More information about the llvm-commits
mailing list