[llvm] [LAA] Add stencil group merging to reduce runtime pointer checks (PR #187252)
David Sherwood via llvm-commits
llvm-commits at lists.llvm.org
Fri Jun 12 06:39:44 PDT 2026
================
@@ -737,6 +763,415 @@ void RuntimePointerChecking::groupChecks(
}
}
+/// Result of decomposing a SCEV expression into stencil offset form:
+/// Offset = Constant + sum(Coefficients[stride] * stride)
+/// where each stride is a loop-invariant SCEV expression.
+struct StencilDecomposition {
+ int64_t Constant = 0;
+ /// Map from loop-invariant stride SCEV to its integer coefficient.
+ SmallMapVector<const SCEV *, int64_t, 4> Coefficients;
+};
+
+/// Try to decompose \p Expr into a stencil offset function of loop-invariant
+/// strides: C + a1*s1 + a2*s2 + ...
+/// Relies on SCEV's canonical form: AddExpr operands are flattened (N-ary),
+/// MulExpr has the constant operand first when present.
+/// Returns std::nullopt if the expression contains non-stencil terms or any
+/// SCEV constant doesn't fit in int64_t (we commit to the signed
+/// interpretation; values that need more than 64 significant bits are
+/// out of scope).
+static std::optional<StencilDecomposition>
+decomposeStencilOffset(const SCEV *Expr, ScalarEvolution &SE, const Loop &L) {
+ StencilDecomposition D;
+
+ // The only caller passes a difference of two pointer Starts, and a Start is
+ // always loop-invariant (getStartAndEndForAccess asserts it). So Expr is
+ // loop-invariant and every term below is loop-invariant too.
+ assert(SE.isLoopInvariant(Expr, &L) && "expected a loop-invariant offset");
+
+ // Collect top-level additive terms.
+ SmallVector<const SCEV *, 4> Terms;
+ if (auto *Add = dyn_cast<SCEVAddExpr>(Expr))
+ append_range(Terms, Add->operands());
+ else
+ Terms.push_back(Expr);
+
+ const SCEVConstant *C;
+ const SCEV *Stride;
+ for (const SCEV *Term : Terms) {
+ if (match(Term, m_SCEVConstant(C))) {
+ auto V = C->getAPInt().trySExtValue();
+ if (!V)
+ return std::nullopt;
+ D.Constant += *V;
+ } else if (match(Term, m_scev_Mul(m_SCEVConstant(C), m_SCEV(Stride)))) {
+ assert(SE.isLoopInvariant(Stride, &L) && "stride must be loop-invariant");
+ auto V = C->getAPInt().trySExtValue();
+ if (!V)
+ return std::nullopt;
+ D.Coefficients[Stride] += *V;
+ } else {
+ assert(SE.isLoopInvariant(Term, &L) && "term must be loop-invariant");
+ D.Coefficients[Term] += 1;
+ }
+ }
+ return D;
+}
+
+void RuntimePointerChecking::mergeStencilGroups(PredicatedScalarEvolution &PSE,
+ Loop &L) {
+ LLVM_DEBUG(dbgs() << "LAA: Attempting stencil group merging on "
+ << CheckingGroups.size() << " groups\n");
+
+ if (CheckingGroups.size() < 2)
+ return;
+
+ // We try to merge groups produced by groupChecks when their pointers follow
+ // a stencil access pattern. groupChecks intentionally only merges pointers
+ // whose min/max bounds differ by constants, because that keeps each runtime
+ // check precise. Stencil kernels often read the same underlying object at
+ // several loop-invariant stride offsets, so those groups remain separate and
+ // can produce too many checks. Here we trade some precision for fewer
+ // checks by replacing a set of same-dependence, same-alias read groups with
+ // one conservative bounding group.
+ //
+ // We use the following algorithm to construct a merged stencil group:
+ // - collect checking groups that share both DependencySetId and AliasSetId;
+ // - reject groups with writes, predicated accesses, different access
+ // ranges, or different recurrence steps;
+ // - use one member as the base and decompose each other member's offset
+ // from that base as C + sum(Coeff[Stride] * Stride), where Stride is
+ // loop-invariant;
+ // - build one bounding range by taking the min/max constant offset and the
+ // min/max coefficient for each stride, adding predicates for strides
+ // that are not already known positive;
+ // - commit the merge only if the local cost model reduces the number of
+ // checks after accounting for any new predicates.
+
+ // Stencil merging runs when either:
+ // - the flag is set to 'force' (-stencil-runtime-check-merge=force), or
+ // - the flag is set to 'auto' (-stencil-runtime-check-merge=auto) AND the
+ // current check count exceeds the auto-trigger threshold, where the
+ // vectorizer would otherwise reject the loop for having too many runtime
+ // checks. In that case the merge can only improve things: at worst we
+ // decline to merge and behave as before.
+ if (StencilMerge == StencilMergePolicy::Off) {
+ LLVM_DEBUG(dbgs() << "LAA: stencil merge disabled\n");
+ return;
+ }
+
+ if (StencilMerge == StencilMergePolicy::Auto) {
+ unsigned TotalChecks = 0;
+ for (unsigned I = 0; I < CheckingGroups.size(); ++I)
+ for (unsigned J = I + 1; J < CheckingGroups.size(); ++J)
+ if (needsChecking(CheckingGroups[I], CheckingGroups[J]))
+ ++TotalChecks;
+
+ if (TotalChecks <= StencilMergeCheckThreshold) {
+ LLVM_DEBUG(dbgs() << "LAA: " << TotalChecks
+ << " checks <= threshold, skipping stencil merge\n");
+ return;
+ }
+ LLVM_DEBUG(
+ dbgs() << "LAA: " << TotalChecks
+ << " checks > threshold, proceeding with stencil merge\n");
+ } else {
+ LLVM_DEBUG(dbgs() << "LAA: stencil merge forced via flag\n");
+ }
+
+ // Group CheckingGroups by (DependencySetId, AliasSetId) pair.
+ // DependencySetId alone is not unique: it resets per alias set, so
+ // pointers in different alias sets can share the same DependencySetId.
+ // Use MapVector for deterministic iteration order across platforms.
+ using DepAliasKey = std::pair<unsigned, unsigned>;
+ MapVector<DepAliasKey, SmallVector<unsigned, 4>> DepSetToGroups;
+ for (unsigned I = 0; I < CheckingGroups.size(); ++I) {
+ const auto &P = Pointers[CheckingGroups[I].Members[0]];
+ DepSetToGroups[{P.DependencySetId, P.AliasSetId}].push_back(I);
+ }
+
+ SmallDenseSet<unsigned, 4> MergedGroupIndices;
+ SmallVector<RuntimeCheckingPtrGroup, 2> NewMergedGroups;
+ // Track strides that already have committed predicates (across all DepSets).
+ SmallDenseSet<const SCEV *, 4> CommittedStridePredicates;
+
+ for (auto &[DepAliasKey, GroupIndices] : DepSetToGroups) {
+ [[maybe_unused]] auto [DepId, ASId] = DepAliasKey;
+ if (GroupIndices.size() < 2)
+ continue;
+
+ // Collect all member pointers across these groups.
+ SmallVector<unsigned, 8> AllMembers;
+ for (unsigned GI : GroupIndices)
+ append_range(AllMembers, CheckingGroups[GI].Members);
+
+ // Only merge read-only groups. Stencil patterns read an array at
+ // multiple offsets and write to a different array (different DepSet).
+ // Mixing reads and writes within a merged group complicates the cost
+ // model and doesn't match known stencil patterns.
+ if (any_of(AllMembers,
+ [&](unsigned Idx) { return Pointers[Idx].IsWritePtr; })) {
+ LLVM_DEBUG(dbgs() << "LAA: Skipping DepSet(" << DepId << "," << ASId
+ << ") with write access\n");
+ continue;
+ }
+
+ // We do not allow predicated accesses. They may result in overestimation
+ // of the boundaries. Imagine a stencil access where we must skip some first
+ // or last iterations because the stencil does not fit the array and has to
+ // go from 1..N-2 although the array is [0..N-1] (for example the dilate
+ // kernel from llvm-test-suite ImageProcessing/Dilate, which reads the
+ // neighbours of every pixel and guards the borders with conditions).
+ // Merging such bounds would widen the already overestimated range further.
+ // This is not necessary in StencilMergePolicy::Auto mode, but skipping it
+ // in StencilMergePolicy::Force mode causes a regression on that benchmark.
+ //
+ // Look at the block of the actual load/store, not of the pointer: a
+ // loop-invariant address is computed in the preheader, outside the loop.
+ if (any_of(AllMembers, [&](unsigned Idx) {
+ const PointerInfo &P = Pointers[Idx];
+ return any_of(
+ DC.getInstructionsForAccess(P.PointerValue, P.IsWritePtr),
+ [&](Instruction *I) {
+ return LoopAccessInfo::blockNeedsPredication(I->getParent(), &L,
+ DC.getDT());
+ });
+ })) {
+ LLVM_DEBUG(dbgs() << "LAA: Skipping DepSet(" << DepId << "," << ASId
+ << ") with predicated access\n");
+ continue;
+ }
+
+ LLVM_DEBUG(dbgs() << "LAA: Analyzing DepSet(" << DepId << "," << ASId
+ << ") with " << AllMembers.size() << " members, base: "
+ << *Pointers[AllMembers[0]].Start << "\n");
+
+ // Use the first member as the reference for decomposition. All offsets
+ // are computed relative to BaseLow, and MergedHigh is built from
+ // BaseHigh (see merged-bounds computation below).
+ const SCEV *BaseLow = Pointers[AllMembers[0]].Start;
+ const SCEV *BaseHigh = Pointers[AllMembers[0]].End;
+
+ const SCEV *BaseStep = nullptr;
+ if (const auto *AR = dyn_cast<SCEVAddRecExpr>(Pointers[AllMembers[0]].Expr))
+ if (AR->getLoop() == &L)
+ BaseStep = AR->getStepRecurrence(*SE);
+
+ if (!BaseStep)
+ continue;
+
+ // Verify all members have the same access range (End - Start).
+ // MergedHigh = BaseHigh + max_offsets is only correct when every member's
+ // range equals BaseHigh - BaseLow. Compute the ranges first and compare
+ // them by subtracting, so algebraically equal but non-identical SCEVs still
+ // match. Bail out if any subtraction produces SCEVCouldNotCompute.
+ const SCEV *BaseRange = SE->getMinusSCEV(BaseHigh, BaseLow);
+ if (isa<SCEVCouldNotCompute>(BaseRange)) {
+ LLVM_DEBUG(dbgs() << "LAA: Base access range not computable, "
+ "skipping DepSet\n");
+ continue;
+ }
+ if (any_of(AllMembers, [&](unsigned Idx) {
+ const SCEV *Range =
+ SE->getMinusSCEV(Pointers[Idx].End, Pointers[Idx].Start);
+ if (isa<SCEVCouldNotCompute>(Range))
+ return true;
+ if (Range == BaseRange)
+ return false;
+ const SCEV *RangeDiff = SE->getMinusSCEV(Range, BaseRange);
+ return isa<SCEVCouldNotCompute>(RangeDiff) || !RangeDiff->isZero();
+ })) {
+ LLVM_DEBUG(
+ dbgs() << "LAA: Member with different or not computable access "
+ "range, skipping DepSet\n");
+ continue;
+ }
+
+ // Require all members to have the same recurrence step. Equal ranges
+ // (checked above) are what the merged bounds actually need, and a different
+ // step usually means a different range. But ranges can be equal by accident
+ // - e.g. an invariant access whose range matches the stride, or a loop with
+ // a single iteration. The base member is picked arbitrarily, so together
+ // with the BaseStep check above this keeps the decision the same no matter
+ // which member comes first: we only merge recurrences with one common step.
+ if (any_of(AllMembers, [&](unsigned Idx) {
+ const SCEV *Step = nullptr;
+ if (const auto *AR = dyn_cast<SCEVAddRecExpr>(Pointers[Idx].Expr))
+ if (AR->getLoop() == &L)
+ Step = AR->getStepRecurrence(*SE);
+ return Step != BaseStep;
+ })) {
+ LLVM_DEBUG(dbgs() << "LAA: Member with different step, "
+ "skipping DepSet\n");
+ continue;
+ }
+ SmallDenseMap<const SCEV *, int64_t, 4> MinCoeff, MaxCoeff;
+ int64_t CMin = 0, CMax = 0;
+ SmallSetVector<const SCEV *, 4> LocalStridesNeedingPreds;
+
+ // Decompose one member's offset (relative to BaseLow) and fold its
+ // constant and per-stride coefficients into the running bounding box.
+ // Returns false if the offset is not in stencil form (so the whole
+ // DepSet is skipped).
+ const auto AccumulateOffset = [&](unsigned Idx) -> bool {
+ const SCEV *LowOffset = SE->getMinusSCEV(Pointers[Idx].Start, BaseLow);
+ if (isa<SCEVCouldNotCompute>(LowOffset))
+ return false;
+ auto DLow = decomposeStencilOffset(LowOffset, *SE, L);
+ if (!DLow) {
+ LLVM_DEBUG(dbgs() << "LAA: Member " << Idx
+ << " NOT decomposable: " << *LowOffset << "\n");
+ return false;
+ }
+
+ CMin = std::min(CMin, DLow->Constant);
+ CMax = std::max(CMax, DLow->Constant);
+
+ for (const auto &[Stride, Coeff] : DLow->Coefficients) {
+ auto MinIt = MinCoeff.find(Stride);
+ if (MinIt == MinCoeff.end()) {
+ MinCoeff[Stride] = Coeff;
+ MaxCoeff[Stride] = Coeff;
+ } else {
+ MinIt->second = std::min(MinIt->second, Coeff);
+ MaxCoeff[Stride] = std::max(MaxCoeff[Stride], Coeff);
+ }
+
+ if (!SE->isKnownPositive(Stride))
+ LocalStridesNeedingPreds.insert(Stride);
+ }
+
+ LLVM_DEBUG(dbgs() << "LAA: Member " << Idx << ": C=" << DLow->Constant
+ << ", strides=" << DLow->Coefficients.size() << "\n");
+ return true;
+ };
+
+ if (!all_of(AllMembers, AccumulateOffset))
+ continue;
+
+ // The base member (offset = 0) implicitly has coefficient 0 for every
+ // stride. Members that lack a particular stride term also have an
+ // implicit coefficient of 0 for that stride. Clamp so that the
+ // bounding-box always includes coefficient 0, which is guaranteed to
+ // exist (at least from the base member).
+ for (auto &[Stride, MinC] : MinCoeff)
----------------
david-arm wrote:
nit: Maybe do
```
for (auto &[_, MinC] : MinCoeff)
MinC = std::min(MinC, (int64_t)0);
```
in each case, since Stride is unused.
https://github.com/llvm/llvm-project/pull/187252
More information about the llvm-commits
mailing list