[llvm] [LoopVectorize] - Add tighter runtime memory check threshold for inner loops. (PR #219489)
Pawan Nirpal via llvm-commits
llvm-commits at lists.llvm.org
Fri Aug 28 07:40:57 PDT 2026
https://github.com/pawan-nirpal-031 updated https://github.com/llvm/llvm-project/pull/219489
>From 975baa2be363a52f6096692f322254cb782f0cc8 Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Fri, 28 Aug 2026 07:03:53 -0700
Subject: [PATCH] [LoopVectorize] - Add tighter runtime memory check threshold
for inner loops
---
.../Transforms/Vectorize/LoopVectorize.cpp | 97 ++++++++++++-------
1 file changed, 63 insertions(+), 34 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 1f153afd6bedd..49c1ea0e41fb2 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -203,6 +203,19 @@ static cl::opt<unsigned> VectorizeMemoryCheckThreshold(
"vectorize-memory-check-threshold", cl::init(128), cl::Hidden,
cl::desc("The maximum allowed number of runtime memory checks"));
+// When vectorizing an inner loop, runtime memory overlap checks execute on
+// every iteration of the enclosing outer loop. If the vectorizer's outer-loop
+// interleaving creates a large number of pointer groups (case in point: from
+// multiple memory ptrs derived from the same base ptr), the resulting O(N^2)
+// pairwise checks become a significant overhead that the existing cost model
+// does not account for. This threshold applies a tighter limit on the number of
+// runtime pointer checks for inner loops. The existing global threshold (128)
+// remains unchanged for top-level loops where checks execute only once.
+static cl::opt<unsigned> VectorizeMemoryCheckInnerLoopThreshold(
+ "vectorize-memory-check-inner-loop-threshold", cl::init(12), cl::Hidden,
+ cl::desc("The maximum allowed number of runtime memory checks for inner "
+ "loops whose checks are not hoistable out of an outer loop"));
+
static cl::opt<bool> ForcePartialAliasingVectorization(
"force-partial-aliasing-vectorization", cl::init(false), cl::Hidden,
cl::desc("Replace pointer diff checks with alias masks."));
@@ -264,7 +277,8 @@ static cl::opt<bool> EnableInterleavedMemAccesses(
/// predication, or in order to mask away gaps.
static cl::opt<bool> EnableMaskedInterleavedMemAccesses(
"enable-masked-interleaved-mem-accesses", cl::init(false), cl::Hidden,
- cl::desc("Enable vectorization on masked interleaved memory accesses in a loop"));
+ cl::desc("Enable vectorization on masked interleaved memory accesses in a "
+ "loop"));
static cl::opt<unsigned> ForceTargetNumScalarRegs(
"force-target-num-scalar-regs", cl::init(0), cl::Hidden,
@@ -1347,10 +1361,9 @@ class LoopVectorizationCostModel {
InstructionCost getConsecutiveMemOpCost(Instruction *I, ElementCount VF,
InstWidening Kind);
- /// The cost calculation for Load/Store instruction \p I with uniform pointer -
- /// Load: scalar load + broadcast.
- /// Store: scalar store + (loop invariant value stored? 0 : extract of last
- /// element)
+ /// The cost calculation for Load/Store instruction \p I with uniform pointer
+ /// - Load: scalar load + broadcast. Store: scalar store + (loop invariant
+ /// value stored? 0 : extract of last element)
InstructionCost getUniformMemOpCost(Instruction *I, ElementCount VF) const;
/// Estimate the overhead of scalarizing an instruction. This is a
@@ -1588,8 +1601,20 @@ class GeneratedRTChecks {
// runtime checks needs to be generated.
// TODO: Skip cutoff if the loop is guaranteed to execute, e.g. due to
// profile info.
- CostTooHigh =
- LAI.getNumRuntimePointerChecks() > VectorizeMemoryCheckThreshold;
+ unsigned NumChecks = LAI.getNumRuntimePointerChecks();
+ unsigned EffectiveThreshold = VectorizeMemoryCheckThreshold;
+
+ // For inner loops, apply a tighter threshold. When vectorizing an inner
+ // loop, runtime memory overlap checks execute on every iteration of the
+ // enclosing outer loop. If the vectorizer's outer-loop interleaving creates
+ // a large number of pointer groups (from multiple memory ptrs derived from
+ // the same base ptr), the resulting O(N^2) pairwise checks become a
+ // significant overhead that the existing cost model does not account for.
+ if (L->getParentLoop() &&
+ NumChecks > VectorizeMemoryCheckInnerLoopThreshold)
+ EffectiveThreshold = VectorizeMemoryCheckInnerLoopThreshold;
+
+ CostTooHigh = NumChecks > EffectiveThreshold;
if (CostTooHigh) {
// Mark runtime checks as never succeeding when they exceed the threshold.
MemRuntimeCheckCond = ConstantInt::getTrue(L->getHeader()->getContext());
@@ -1598,9 +1623,13 @@ class GeneratedRTChecks {
return OptimizationRemarkAnalysisAliasing(
DEBUG_TYPE, "TooManyMemoryRuntimeChecks", L->getStartLoc(),
L->getHeader())
- << "loop not vectorized: too many memory checks needed";
+ << "loop not vectorized: too many memory checks needed"
+ << (NumChecks > VectorizeMemoryCheckThreshold
+ ? ""
+ : " (inner loop threshold exceeded)");
});
- LLVM_DEBUG(dbgs() << "LV: Too many memory checks needed.\n");
+ LLVM_DEBUG(dbgs() << "LV: Too many memory checks needed (" << NumChecks
+ << " > " << EffectiveThreshold << ").\n");
return;
}
@@ -1908,12 +1937,13 @@ static void collectSupportedLoops(Loop &L, LoopInfo *LI,
//===----------------------------------------------------------------------===//
/// For the given VF and UF and maximum trip count computed for the loop, return
-/// whether the induction variable might overflow in the vectorized loop. If not,
-/// then we know a runtime overflow check always evaluates to false and can be
-/// removed.
-static bool isIndvarOverflowCheckKnownFalse(
- const LoopVectorizationCostModel *Cost,
- ElementCount VF, std::optional<unsigned> UF = std::nullopt) {
+/// whether the induction variable might overflow in the vectorized loop. If
+/// not, then we know a runtime overflow check always evaluates to false and can
+/// be removed.
+static bool
+isIndvarOverflowCheckKnownFalse(const LoopVectorizationCostModel *Cost,
+ ElementCount VF,
+ std::optional<unsigned> UF = std::nullopt) {
// Always be conservative if we don't know the exact unroll factor.
unsigned MaxUF = UF ? *UF
: std::max(Cost->TTI.getMaxInterleaveFactor(VF, false),
@@ -2288,7 +2318,8 @@ void LoopVectorizationCostModel::collectLoopScalars(ElementCount VF) {
auto ForcedScalar = ForcedScalars.find(VF);
if (ForcedScalar != ForcedScalars.end())
for (auto *I : ForcedScalar->second) {
- LLVM_DEBUG(dbgs() << "LV: Found (forced) scalar instruction: " << *I << "\n");
+ LLVM_DEBUG(dbgs() << "LV: Found (forced) scalar instruction: " << *I
+ << "\n");
Worklist.insert(I);
}
@@ -2387,7 +2418,7 @@ bool LoopVectorizationCostModel::isScalarWithPredication(Instruction *I,
// Do we have a non-scalar lowering for this predicated
// instruction? No - it is scalar with predication.
- switch(I->getOpcode()) {
+ switch (I->getOpcode()) {
default:
return true;
case Instruction::Call: {
@@ -2446,7 +2477,7 @@ bool LoopVectorizationCostModel::isPredicatedInst(Instruction *I) const {
// having at least one active lane (the first). If the side-effects of the
// instruction are invariant, executing it w/o (the tail-folding) mask is safe
// - it will cause the same side-effects as when masked.
- switch(I->getOpcode()) {
+ switch (I->getOpcode()) {
default:
llvm_unreachable(
"instruction should have been considered by earlier checks");
@@ -2693,8 +2724,8 @@ void LoopVectorizationCostModel::collectLoopUniforms(ElementCount VF) {
// where only a single instance out of VF should be formed.
auto AddToWorklistIfAllowed = [&](Instruction *I) -> void {
if (IsOutOfScope(I)) {
- LLVM_DEBUG(dbgs() << "LV: Found not uniform due to scope: "
- << *I << "\n");
+ LLVM_DEBUG(dbgs() << "LV: Found not uniform due to scope: " << *I
+ << "\n");
return;
}
if (isPredicatedInst(I)) {
@@ -3611,7 +3642,8 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
LoopCost = CM.expectedCost(VF);
else
LoopCost = cost(Plan, VF, &R);
- assert(LoopCost.isValid() && "Expected to have chosen a VF with valid cost");
+ assert(LoopCost.isValid() &&
+ "Expected to have chosen a VF with valid cost");
// Loop body is free and there is no need for interleaving.
if (LoopCost == 0)
@@ -3916,11 +3948,9 @@ bool LoopVectorizationCostModel::useEmulatedMaskMemRefHack(
// from moving "masked load/store" check from legality to cost model.
// Masked Load/Gather emulation was previously never allowed.
// Limited number of Masked Store/Scatter emulation was allowed.
- assert((isPredicatedInst(I)) &&
- "Expecting a scalar emulated instruction");
+ assert((isPredicatedInst(I)) && "Expecting a scalar emulated instruction");
return isa<LoadInst>(I) ||
- (isa<StoreInst>(I) &&
- NumPredStores > NumberOfStoresToPredicate);
+ (isa<StoreInst>(I) && NumPredStores > NumberOfStoresToPredicate);
}
void LoopVectorizationCostModel::collectInstsToScalarize(ElementCount VF) {
@@ -4132,10 +4162,9 @@ InstructionCost LoopVectorizationCostModel::expectedCost(ElementCount VF) {
///
/// This SCEV can be sent to the Target in order to estimate the address
/// calculation cost.
-static const SCEV *getAddressAccessSCEV(
- Value *Ptr,
- PredicatedScalarEvolution &PSE,
- const Loop *TheLoop) {
+static const SCEV *getAddressAccessSCEV(Value *Ptr,
+ PredicatedScalarEvolution &PSE,
+ const Loop *TheLoop) {
const SCEV *Addr = PSE.getSCEV(Ptr);
return vputils::isAddressSCEVForCost(Addr, *PSE.getSE(), TheLoop) ? Addr
: nullptr;
@@ -4609,7 +4638,7 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
for (BasicBlock *BB : TheLoop->blocks()) {
// For each instruction in the old loop.
for (Instruction &I : *BB) {
- Value *Ptr = getLoadStorePointerOperand(&I);
+ Value *Ptr = getLoadStorePointerOperand(&I);
if (!Ptr)
continue;
@@ -4736,7 +4765,7 @@ void LoopVectorizationCostModel::setCostBasedWideningDecision(ElementCount VF) {
for (BasicBlock *BB : TheLoop->blocks())
for (Instruction &I : *BB) {
Instruction *PtrDef =
- dyn_cast_or_null<Instruction>(getLoadStorePointerOperand(&I));
+ dyn_cast_or_null<Instruction>(getLoadStorePointerOperand(&I));
if (PtrDef && TheLoop->contains(PtrDef) &&
getWideningDecision(&I, VF) != CM_GatherScatter)
AddrDefs.insert(PtrDef);
@@ -5103,7 +5132,7 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
const auto [Op1VK, Op1VP] = TTI::getOperandInfo(Op0);
const auto [Op2VK, Op2VP] = TTI::getOperandInfo(Op1);
assert(Op0->getType()->getScalarSizeInBits() == 1 &&
- Op1->getType()->getScalarSizeInBits() == 1);
+ Op1->getType()->getScalarSizeInBits() == 1);
return TTI.getArithmeticInstrCost(
match(I, m_LogicalOr()) ? Instruction::Or : Instruction::And,
@@ -7571,8 +7600,8 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
auto *VPI = dyn_cast<VPInstruction>(R);
return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
};
- auto *RdxResult = cast<VPInstruction>(
- vputils::findRecipe(ReductionPhi->getBackedgeValue(), IsReductionResult));
+ auto *RdxResult = cast<VPInstruction>(vputils::findRecipe(
+ ReductionPhi->getBackedgeValue(), IsReductionResult));
assert(RdxResult && "expected to find reduction result");
VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
More information about the llvm-commits
mailing list