[llvm] [SLP] Refine loop-aware gather cost and admit sibling-loop subtrees (PR #192801)
Ryan Buchner via llvm-commits
llvm-commits at lists.llvm.org
Tue Apr 28 11:54:10 PDT 2026
================
@@ -16063,31 +16133,90 @@ unsigned BoUpSLP::getScaleToLoopIterations(const TreeEntry &TE, Value *Scalar,
} else {
Parent = TE.getMainOp()->getParent();
}
- if (const Loop *L = LI->getLoopFor(Parent)) {
- const auto It = LoopToScaleFactor.find(L);
- if (It != LoopToScaleFactor.end())
- return It->second;
- unsigned Scale = 1;
- if (const Loop *NonInvL = findInnermostNonInvariantLoop(
- L, Scalar ? ArrayRef(Scalar) : ArrayRef(TE.Scalars))) {
- Scale = getLoopTripCount(NonInvL, *SE);
- for (const Loop *LN : getLoopNest(NonInvL)) {
- if (LN == L)
- break;
- auto LNRes = LoopToScaleFactor.try_emplace(LN, 0);
- auto &LoopScale = LNRes.first->getSecond();
- if (!LNRes.second) {
- Scale *= LoopScale;
- break;
- }
- Scale *= getLoopTripCount(LN, *SE);
- LoopScale = Scale;
- }
- }
- LoopToScaleFactor.try_emplace(L, Scale);
- return Scale;
- }
- return 1;
+ const Loop *L = LI->getLoopFor(Parent);
+ if (!L)
+ return 1;
+ // The entry's cost is paid once per execution of the innermost loop in
+ // which some of its operands are variant. Operands that are invariant in
+ // all enclosing loops are executed once (LICM will hoist them out).
+ return getLoopNestScale(findInnermostNonInvariantLoop(
+ L, Scalar ? ArrayRef(Scalar) : ArrayRef(TE.Scalars)));
+}
+
+uint64_t BoUpSLP::getLoopNestScale(const Loop *L) {
+ if (!L || LoopAwareTripCount == 0)
+ return 1;
+ if (auto It = LoopNestScaleCache.find(L); It != LoopNestScaleCache.end())
+ return It->second;
+ // Collect loops from L outward up to (but not including) the first cached
+ // ancestor or the function top, then walk back inward multiplying trip
+ // counts. Use uint64_t to avoid silent overflow on deep/large nests.
+ SmallVector<const Loop *> Chain;
+ for (const Loop *Cur = L; Cur; Cur = Cur->getParentLoop()) {
+ if (LoopNestScaleCache.contains(Cur))
+ break;
+ Chain.push_back(Cur);
+ }
+ assert(!Chain.empty() && "Early-return above should have handled cache hit.");
+ uint64_t Scale = 1;
+ if (const Loop *Parent = Chain.back()->getParentLoop())
+ Scale = LoopNestScaleCache.lookup(Parent);
+ // Walk from the outermost uncached loop inward, accumulating trip counts.
+ // Use SaturatingMultiply to clamp at uint64_t max on deep/large nests
+ // rather than wrapping around.
+ for (const Loop *Cur : reverse(Chain)) {
+ uint64_t TC = std::max<uint64_t>(1, getLoopTripCount(Cur, *SE));
+ Scale = SaturatingMultiply(Scale, TC);
+ LoopNestScaleCache.try_emplace(Cur, std::max<uint64_t>(1, Scale));
+ }
+ return std::max<uint64_t>(1, Scale);
+}
+
+uint64_t BoUpSLP::getGatherNodeEffectiveScale(const TreeEntry &TE) {
+ // Only meaningful for gather/buildvector-like entries; the per-lane
+ // insertelements that make up such an entry are LICM-hoistable by
+ // optimizeGatherSequence() when their operand is loop-invariant.
+ assert((TE.isGather() || TE.State == TreeEntry::SplitVectorize) &&
+ "Expected gather/split tree entry.");
+
+ uint64_t BaseScale = getScaleToLoopIterations(TE);
+ if (!PerLaneGatherScale || LoopAwareTripCount == 0 || BaseScale <= 1)
+ return BaseScale;
+
+ // Average the per-lane execution scales: for each lane, reuse the same
+ // scale helper the rest of the cost model uses, but ask it about that
+ // one lane's value. Lanes that are loop-invariant in the current nest
+ // collapse to their outer-loop scale (or 1 for fully invariant/constant
+ // lanes), which matches the LICM hoisting performed by
+ // optimizeGatherSequence(). Cap per-lane contributions by BaseScale so a
+ // refinement can never raise the cost above the whole-entry scale.
+ // Each lane contributes at most BaseScale, so Sum is bounded above by
+ // N * BaseScale. If BaseScale is near uint64_t max (saturated by
+ // getLoopNestScale on a deep nest) Sum can still overflow uint64_t,
+ // which would silently wrap and produce a wrong average. Use
+ // SaturatingAdd and bail out to BaseScale on overflow: the true average
+ // is bounded above by BaseScale anyway, so this preserves the
+ // refinement's invariant that it can never raise cost.
+ uint64_t Sum = 0;
+ unsigned N = 0;
+ bool Overflow = false;
+ for (Value *V : TE.Scalars) {
+ if (isConstant(V))
+ continue;
+ ++N;
+ uint64_t LaneScale = std::min(getScaleToLoopIterations(TE, V), BaseScale);
+ Sum = SaturatingAdd(Sum, LaneScale, &Overflow);
+ if (Overflow)
+ return BaseScale;
+ }
+ if (N == 0)
+ return BaseScale;
----------------
bababuck wrote:
I thought (probably incorrectly) that we don't create `TreeEntry`'s for all constant nodes.
https://github.com/llvm/llvm-project/pull/192801
More information about the llvm-commits
mailing list