[llvm] [LAA] Allow vectorizing `A[NonZeroNonConstantStride*I] += 1` (PR #186262)
Andrei Elovikov via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 13:52:31 PDT 2026
https://github.com/eas updated https://github.com/llvm/llvm-project/pull/186262
>From cc4bfd1c0b1c46d226ff8a7835b0ae8a80355747 Mon Sep 17 00:00:00 2001
From: Andrei Elovikov <andrei.elovikov at sifive.com>
Date: Wed, 11 Mar 2026 12:45:41 -0700
Subject: [PATCH 1/5] [NFC][LAA] Add a test for a single strided-read-write
access
---
.../single_strided_readwrite.ll | 243 ++++++++++++++++++
1 file changed, 243 insertions(+)
create mode 100644 llvm/test/Analysis/LoopAccessAnalysis/single_strided_readwrite.ll
diff --git a/llvm/test/Analysis/LoopAccessAnalysis/single_strided_readwrite.ll b/llvm/test/Analysis/LoopAccessAnalysis/single_strided_readwrite.ll
new file mode 100644
index 0000000000000..390e694c0b340
--- /dev/null
+++ b/llvm/test/Analysis/LoopAccessAnalysis/single_strided_readwrite.ll
@@ -0,0 +1,243 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes='print<access-info>' -disable-output < %s -enable-mem-access-versioning=false 2>&1 | FileCheck %s
+
+define void @known_safe(ptr %p, i8 %a) {
+; CHECK-LABEL: 'known_safe'
+; CHECK-NEXT: header:
+; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
+; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: IndirectUnsafe:
+; CHECK-NEXT: %ld = load i64, ptr %gep, align 4 ->
+; CHECK-NEXT: store i64 %add, ptr %gep, align 4
+; CHECK-EMPTY:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Grouped accesses:
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+entry:
+ %a.zext = zext i8 %a to i64
+ %stride = add i64 %a.zext, 1
+ br label %header
+
+header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %header ]
+ %iv.next = add nsw i64 %iv, 1
+ %idx = mul nsw nuw i64 %iv, %stride
+
+ %gep = getelementptr inbounds i64, ptr %p, i64 %idx
+ %ld = load i64, ptr %gep
+ %add = add i64 %ld, %iv
+ store i64 %add, ptr %gep
+
+ %exitcond = icmp slt i64 %iv.next, 128
+ br i1 %exitcond, label %header, label %exit
+
+exit:
+ ret void
+}
+
+define void @known_safe_byte_gep(ptr %p, i8 %a) {
+; CHECK-LABEL: 'known_safe_byte_gep'
+; CHECK-NEXT: header:
+; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
+; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: IndirectUnsafe:
+; CHECK-NEXT: %ld = load i64, ptr %gep, align 4 ->
+; CHECK-NEXT: store i64 %add, ptr %gep, align 4
+; CHECK-EMPTY:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Grouped accesses:
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+entry:
+ %a.zext = zext i8 %a to i64
+ %stride.elts = add i64 %a.zext, 1
+ %stride = mul i64 %stride.elts, 8
+ br label %header
+
+header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %header ]
+ %iv.next = add nsw i64 %iv, 1
+ %idx = mul nsw nuw i64 %iv, %stride
+
+ %gep = getelementptr inbounds i8, ptr %p, i64 %idx
+ %ld = load i64, ptr %gep
+ %add = add i64 %ld, %iv
+ store i64 %add, ptr %gep
+
+ %exitcond = icmp slt i64 %iv.next, 128
+ br i1 %exitcond, label %header, label %exit
+
+exit:
+ ret void
+}
+
+; This would require `%a u> 0` RT check.
+define void @known_non_negative(ptr %p, i8 %a) {
+; CHECK-LABEL: 'known_non_negative'
+; CHECK-NEXT: header:
+; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
+; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: IndirectUnsafe:
+; CHECK-NEXT: %ld = load i64, ptr %gep, align 4 ->
+; CHECK-NEXT: store i64 %add, ptr %gep, align 4
+; CHECK-EMPTY:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Grouped accesses:
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+entry:
+ %stride = zext i8 %a to i64
+ br label %header
+
+header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %header ]
+ %iv.next = add nsw i64 %iv, 1
+ %idx = mul nsw nuw i64 %iv, %stride
+
+ %gep = getelementptr inbounds i64, ptr %p, i64 %idx
+ %ld = load i64, ptr %gep
+ %add = add i64 %ld, %iv
+ store i64 %add, ptr %gep
+
+ %exitcond = icmp slt i64 %iv.next, 128
+ br i1 %exitcond, label %header, label %exit
+
+exit:
+ ret void
+}
+
+; This would require `%a u> 0` RT check.
+define void @known_non_negative_scaled_for_byte_gep(ptr %p, i8 %a) {
+; CHECK-LABEL: 'known_non_negative_scaled_for_byte_gep'
+; CHECK-NEXT: header:
+; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
+; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: IndirectUnsafe:
+; CHECK-NEXT: %ld = load i64, ptr %gep, align 4 ->
+; CHECK-NEXT: store i64 %add, ptr %gep, align 4
+; CHECK-EMPTY:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Grouped accesses:
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+entry:
+ %a.zext = zext i8 %a to i64
+ %stride = mul nsw nuw i64 %a.zext, 8
+ br label %header
+
+header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %header ]
+ %iv.next = add nsw i64 %iv, 1
+ %idx = mul nsw nuw i64 %iv, %stride
+
+ %gep = getelementptr inbounds i8, ptr %p, i64 %idx
+ %ld = load i64, ptr %gep
+ %add = add i64 %ld, %iv
+ store i64 %add, ptr %gep
+
+ %exitcond = icmp slt i64 %iv.next, 128
+ br i1 %exitcond, label %header, label %exit
+
+exit:
+ ret void
+}
+
+; This would require `%a u> 8` RT check.
+define void @known_non_negative_nonscaled_for_byte_gep(ptr %p, i8 %a) {
+; CHECK-LABEL: 'known_non_negative_nonscaled_for_byte_gep'
+; CHECK-NEXT: header:
+; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
+; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: IndirectUnsafe:
+; CHECK-NEXT: %ld = load i64, ptr %gep, align 4 ->
+; CHECK-NEXT: store i64 %add, ptr %gep, align 4
+; CHECK-EMPTY:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Grouped accesses:
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+entry:
+ %stride = zext i8 %a to i64
+ br label %header
+
+header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %header ]
+ %iv.next = add nsw i64 %iv, 1
+ %idx = mul nsw nuw i64 %iv, %stride
+
+ %gep = getelementptr inbounds i8, ptr %p, i64 %idx
+ %ld = load i64, ptr %gep
+ %add = add i64 %ld, %iv
+ store i64 %add, ptr %gep
+
+ %exitcond = icmp slt i64 %iv.next, 128
+ br i1 %exitcond, label %header, label %exit
+
+exit:
+ ret void
+}
+
+; This would require `abs(%a) u> 8` RT check.
+define void @arbitrary(ptr %p, i64 %stride) {
+; CHECK-LABEL: 'arbitrary'
+; CHECK-NEXT: header:
+; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
+; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: IndirectUnsafe:
+; CHECK-NEXT: %ld = load i64, ptr %gep, align 4 ->
+; CHECK-NEXT: store i64 %add, ptr %gep, align 4
+; CHECK-EMPTY:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Grouped accesses:
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+entry:
+ br label %header
+
+header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %header ]
+ %iv.next = add nsw i64 %iv, 1
+ %idx = mul nsw nuw i64 %iv, %stride
+
+ %gep = getelementptr inbounds i8, ptr %p, i64 %idx
+ %ld = load i64, ptr %gep
+ %add = add i64 %ld, %iv
+ store i64 %add, ptr %gep
+
+ %exitcond = icmp slt i64 %iv.next, 128
+ br i1 %exitcond, label %header, label %exit
+
+exit:
+ ret void
+}
>From ccc678b64560fa2741093185c3f73b7604a07117 Mon Sep 17 00:00:00 2001
From: Andrei Elovikov <andrei.elovikov at sifive.com>
Date: Thu, 12 Mar 2026 16:11:23 -0700
Subject: [PATCH 2/5] [LAA] Add `getPtrStrideScev` function (NFC'ish)
Not completely NFC because old code gives up on huge stride in bytes
while the new one does that on huge stride in elements (after dividing
stride in bytes by the access size).
---
.../llvm/Analysis/LoopAccessAnalysis.h | 16 +--
llvm/lib/Analysis/LoopAccessAnalysis.cpp | 100 +++++++++++-------
.../Transforms/Vectorize/VPlanTransforms.cpp | 7 +-
.../bounded-access-pattern.ll | 7 +-
4 files changed, 78 insertions(+), 52 deletions(-)
diff --git a/llvm/include/llvm/Analysis/LoopAccessAnalysis.h b/llvm/include/llvm/Analysis/LoopAccessAnalysis.h
index 392321448c895..dbcffb2bb119b 100644
--- a/llvm/include/llvm/Analysis/LoopAccessAnalysis.h
+++ b/llvm/include/llvm/Analysis/LoopAccessAnalysis.h
@@ -879,13 +879,15 @@ replaceSymbolicStrideSCEV(PredicatedScalarEvolution &PSE,
const DenseMap<Value *, const SCEV *> &PtrToStride,
Value *Ptr);
-/// If \p AR is an affine AddRec for \p Lp with a constant step, return the
-/// step in units of \p AccessTy's allocation size. Returns std::nullopt if the
-/// step is not constant, does not divide the access size, or \p AccessTy is a
-/// scalable vector. \p Ptr is only used for debug output and may be null.
-LLVM_ABI std::optional<int64_t>
-getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp, Type *AccessTy,
- Value *Ptr, PredicatedScalarEvolution &PSE);
+/// If \p AR is an affine AddRec for \p Lp with a loop invariant step, return
+/// the step in units of \p AccessTy's allocation size. Returns nullptr if the
+/// step is not loop invariant, does not have statically known direction (sign),
+/// does not divide the access size, or \p AccessTy is a scalable vector. \p Ptr
+/// is only used for debug output and may be null.
+LLVM_ABI const SCEV *getStrideFromAddRec(const SCEVAddRecExpr *AR,
+ const Loop *Lp, Type *AccessTy,
+ Value *Ptr,
+ PredicatedScalarEvolution &PSE);
/// If the pointer has a constant stride return it in units of the access type
/// size. If the pointer is loop-invariant, return 0. Otherwise return
diff --git a/llvm/lib/Analysis/LoopAccessAnalysis.cpp b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
index a5aba968f4ea9..432d287558ef8 100644
--- a/llvm/lib/Analysis/LoopAccessAnalysis.cpp
+++ b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
@@ -967,54 +967,56 @@ class AccessAnalysis {
} // end anonymous namespace
-std::optional<int64_t>
-llvm::getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp,
- Type *AccessTy, Value *Ptr,
- PredicatedScalarEvolution &PSE) {
+const SCEV *llvm::getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp,
+ Type *AccessTy, Value *Ptr,
+ PredicatedScalarEvolution &PSE) {
if (isa<ScalableVectorType>(AccessTy)) {
LLVM_DEBUG(dbgs() << "LAA: Bad stride - Scalable object: " << *AccessTy
<< "\n");
- return std::nullopt;
+ return nullptr;
}
- // The access function must stride over the innermost loop.
- if (Lp != AR->getLoop()) {
+ auto BadStride = [&](auto Str) {
LLVM_DEBUG({
- dbgs() << "LAA: Bad stride - Not striding over innermost loop ";
+ dbgs() << "LAA: Bad stride - " << Str << " ";
if (Ptr)
dbgs() << *Ptr << " ";
dbgs() << "SCEV: " << *AR << "\n";
});
- return std::nullopt;
- }
+ return nullptr;
+ };
+
+ // The access function must stride over the innermost loop.
+ if (Lp != AR->getLoop())
+ return BadStride("Not striding over innermost loop");
+
+ // Check the step is loop invariant.
+ if (!AR->isAffine())
+ return nullptr;
- // Check the step is constant.
const SCEV *Step = AR->getStepRecurrence(*PSE.getSE());
- // Calculate the pointer stride and check if it is constant.
- const APInt *APStepVal;
- if (!match(Step, m_scev_APInt(APStepVal))) {
- LLVM_DEBUG({
- dbgs() << "LAA: Bad stride - Not a constant strided ";
- if (Ptr)
- dbgs() << *Ptr << " ";
- dbgs() << "SCEV: " << *AR << "\n";
- });
- return std::nullopt;
- }
+ auto *SE = PSE.getSE();
+ const SCEV *AbsStep = SE->getAbsExpr(Step, false);
- const auto &DL = Lp->getHeader()->getDataLayout();
- TypeSize AllocSize = DL.getTypeAllocSize(AccessTy);
- int64_t Size = AllocSize.getFixedValue();
+ const SCEV *TypeSizeScev = SE->getSizeOfExpr(
+ Step->getType(), SE->getDataLayout().getTypeAllocSize(AccessTy));
- // Huge step value - give up.
- std::optional<int64_t> StepVal = APStepVal->trySExtValue();
- if (!StepVal)
- return std::nullopt;
+ if (!SE->getURemExpr(AbsStep, TypeSizeScev)->isZero())
+ return BadStride("Not a multiple of access size");
- // Strided access.
- return *StepVal % Size ? std::nullopt : std::make_optional(*StepVal / Size);
+ // There is no ScalarEvolution::getSDiv, emulate that via AbsStep/TypeSize
+ // if the Step sign is known statically.
+ if (!(SE->isKnownNonPositive(Step) || SE->isKnownNonNegative(Step)))
+ return BadStride("Unknown sign");
+
+ const SCEV *AbsStepInElements = SE->getUDivExpr(AbsStep, TypeSizeScev);
+ const SCEV *StepInElements = SE->isKnownNonNegative(Step)
+ ? AbsStepInElements
+ : SE->getNegativeSCEV(AbsStepInElements);
+
+ return StepInElements;
}
/// Check whether \p AR is a non-wrapping AddRec. If \p Ptr is not nullptr, use
@@ -1023,7 +1025,7 @@ llvm::getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp,
static bool
isNoWrap(PredicatedScalarEvolution &PSE, const SCEVAddRecExpr *AR, Value *Ptr,
Type *AccessTy, const Loop *L, const DominatorTree &DT,
- std::optional<int64_t> Stride = std::nullopt,
+ const SCEV *Stride = nullptr,
SmallVectorImpl<const SCEVPredicate *> *Predicates = nullptr) {
// FIXME: This should probably only return true for NUW.
if (any(AR->getNoWrapFlags(SCEV::NoWrapMask)))
@@ -1061,7 +1063,7 @@ isNoWrap(PredicatedScalarEvolution &PSE, const SCEVAddRecExpr *AR, Value *Ptr,
// assumes the object in memory is aligned to the natural alignment.
unsigned AddrSpace = AR->getType()->getPointerAddressSpace();
if (!NullPointerIsDefined(L->getHeader()->getParent(), AddrSpace) &&
- (Stride == 1 || Stride == -1))
+ PSE.getSE()->getAbsExpr(Stride, false)->isOne())
return true;
}
@@ -1322,7 +1324,7 @@ bool AccessAnalysis::createCheckForAccess(
}
if (!isNoWrap(PSE, AR, RTCheckPtrs.size() == 1 ? Ptr : nullptr, AccessTy,
- TheLoop, DT, /*Stride=*/std::nullopt,
+ TheLoop, DT, /*Stride=*/nullptr,
Assume ? &Predicates : nullptr))
return false;
}
@@ -1654,14 +1656,15 @@ void AccessAnalysis::buildDependenceSets() {
}
}
-/// Check whether the access through \p Ptr has a constant stride.
-std::optional<int64_t> llvm::getPtrStride(
+/// Check whether the access through \p Ptr has a loop invariant stride of a
+/// statically known sign.
+static const SCEV *getPtrStrideScev(
PredicatedScalarEvolution &PSE, Type *AccessTy, Value *Ptr, const Loop *Lp,
const DominatorTree &DT, const DenseMap<Value *, const SCEV *> &StridesMap,
bool ShouldCheckWrap, SmallVectorImpl<const SCEVPredicate *> *Predicates) {
const SCEV *PtrScev = replaceSymbolicStrideSCEV(PSE, StridesMap, Ptr);
if (PSE.getSE()->isLoopInvariant(PtrScev, Lp))
- return 0;
+ return PSE.getSE()->getZero(Type::getInt64Ty(AccessTy->getContext()));
assert(Ptr->getType()->isPointerTy() && "Unexpected non-ptr");
@@ -1674,11 +1677,10 @@ std::optional<int64_t> llvm::getPtrStride(
if (!AR) {
LLVM_DEBUG(dbgs() << "LAA: Bad stride - Not an AddRecExpr pointer " << *Ptr
<< " SCEV: " << *PtrScev << "\n");
- return std::nullopt;
+ return nullptr;
}
- std::optional<int64_t> Stride =
- getStrideFromAddRec(AR, Lp, AccessTy, Ptr, PSE);
+ const SCEV *Stride = getStrideFromAddRec(AR, Lp, AccessTy, Ptr, PSE);
if (!ShouldCheckWrap || !Stride)
return Stride;
@@ -1688,7 +1690,23 @@ std::optional<int64_t> llvm::getPtrStride(
LLVM_DEBUG(
dbgs() << "LAA: Bad stride - Pointer may wrap in the address space "
<< *Ptr << " SCEV: " << *AR << "\n");
- return std::nullopt;
+ return nullptr;
+}
+
+/// Check whether the access through \p Ptr has a constant stride.
+std::optional<int64_t> llvm::getPtrStride(
+ PredicatedScalarEvolution &PSE, Type *AccessTy, Value *Ptr, const Loop *Lp,
+ const DominatorTree &DT, const DenseMap<Value *, const SCEV *> &StridesMap,
+ bool ShouldCheckWrap, SmallVectorImpl<const SCEVPredicate *> *Predicates) {
+ const SCEV *StrideScev = getPtrStrideScev(
+ PSE, AccessTy, Ptr, Lp, DT, StridesMap, ShouldCheckWrap, Predicates);
+ if (!StrideScev)
+ return std::nullopt;
+ const APInt *APStride = nullptr;
+ if (!match(StrideScev, m_scev_APInt(APStride)))
+ return std::nullopt;
+
+ return APStride->trySExtValue();
}
/// Check whether the access through \p Ptr has a constant stride.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 5834f38e96b84..60937340cc08a 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -7150,7 +7150,12 @@ static std::optional<int64_t> getConstantStride(VPValue *Addr, Type *AccessTy,
if (!AddRec)
return {};
- return getStrideFromAddRec(AddRec, L, AccessTy, /*Ptr=*/nullptr, PSE);
+ const auto *Stride = dyn_cast_or_null<SCEVConstant>(
+ getStrideFromAddRec(AddRec, L, AccessTy, /*Ptr=*/nullptr, PSE));
+ if (!Stride)
+ return {};
+
+ return Stride->getAPInt().trySExtValue();
}
void VPlanTransforms::makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range,
diff --git a/llvm/test/Analysis/LoopAccessAnalysis/bounded-access-pattern.ll b/llvm/test/Analysis/LoopAccessAnalysis/bounded-access-pattern.ll
index 020d808f7b503..7e7d4843bc6ce 100644
--- a/llvm/test/Analysis/LoopAccessAnalysis/bounded-access-pattern.ll
+++ b/llvm/test/Analysis/LoopAccessAnalysis/bounded-access-pattern.ll
@@ -1055,15 +1055,16 @@ define void @bounded_mul_huge_scale_as1(ptr addrspace(1) %a) {
; CHECK-EMPTY:
; CHECK-NEXT: Expressions re-written:
; CHECK-NEXT: [PSE] %gep = getelementptr inbounds i8, ptr addrspace(1) %a, i128 %off:
-; CHECK-NEXT: ((36893488147419103232 * (zext i2 {0,+,1}<%loop> to i128))<nuw><nsw> + %a)<nuw>
-; CHECK-NEXT: --> {%a,+,36893488147419103232}<nw><%loop>
+; CHECK-NEXT: ((147573952589676412928 * (zext i2 {0,+,1}<%loop> to i128))<nuw><nsw> + %a)<nuw>
+; CHECK-NEXT: --> {%a,+,147573952589676412928}<nw><%loop>
;
entry:
br label %loop
loop:
%iv = phi i128 [ 0, %entry ], [ %iv.next, %loop ]
%idx = urem i128 %iv, 4
- %off = mul i128 %idx, 36893488147419103232
+ ; Huge enough for the stride in elements (not bytes!) to be out of i64 range:
+ %off = mul i128 %idx, u0x80000000000000000
%gep = getelementptr inbounds i8, ptr addrspace(1) %a, i128 %off
%ld = load i64, ptr addrspace(1) %gep, align 8
%add = add i64 %ld, 1
>From 5ce0a1601c1a0e49d0dfeca6e54fe44082228885 Mon Sep 17 00:00:00 2001
From: Andrei Elovikov <andrei.elovikov at sifive.com>
Date: Wed, 11 Mar 2026 14:47:09 -0700
Subject: [PATCH 3/5] [LAA] Allow vectorizing `A[NonZeroNonConstantStride*I] +=
1`
In this patch only do that when we can statically prove that
non-constant stride is non-zero and the resulting index doesn't
overflow. That can later be extended to introduce run-time check when
not provable in compile-time.
My main motivation for this is to move unit-strideness speculation to a
VPlan-based transformation. However, it cannot be done right now because
sometimes such speculation affects legality and we simply avoid
vectorizing loop if it's not done. As such, we need to extend LAA to
properly support dependence analysis/RT checks for strided access
without speculating for it being one. This PR is expected to be the
first one on that journey.
---
llvm/lib/Analysis/LoopAccessAnalysis.cpp | 25 +++--
.../single_strided_readwrite.ll | 95 ++++++++++++++++---
2 files changed, 102 insertions(+), 18 deletions(-)
diff --git a/llvm/lib/Analysis/LoopAccessAnalysis.cpp b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
index 432d287558ef8..133072b369e89 100644
--- a/llvm/lib/Analysis/LoopAccessAnalysis.cpp
+++ b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
@@ -2792,14 +2792,27 @@ bool LoopAccessInfo::analyzeLoop(AAResults *AA, const LoopInfo *LI,
// allows us to vectorize expressions such as A[i] += x; Because the address
// of A[i] is a read-write pointer. This only works if the index of A[i] is
// strictly monotonic, which we approximate (conservatively) via
- // getPtrStride. If the address is unknown (e.g. A[B[i]]) then we may read,
- // modify, and write overlapping words. Note that "zero stride" is unsafe
- // and is being handled below.
+ // getPtrStrideScev. If the address is unknown (e.g. A[B[i]]) then we may
+ // read, modify, and write overlapping words. Note that "zero stride" is
+ // unsafe and is being handled below.
bool IsReadOnlyPtr = false;
Type *AccessTy = getLoadStoreType(LD);
- if (Seen.insert({Ptr, AccessTy}).second ||
- !getPtrStride(*PSE, AccessTy, Ptr, TheLoop, *DT, SymbolicStrides, false,
- true)) {
+ auto IsSafeReadWrite = [&] {
+ const SCEV *Stride = getPtrStrideScev(*PSE, AccessTy, Ptr, TheLoop, *DT,
+ SymbolicStrides, true, nullptr);
+ if (!Stride)
+ return false;
+
+ // Statically known invariant address, preserve old behavior for the
+ // LoopDistributePass. For LoopVectorizer we will detect a load from the
+ // uniform store pointer and bail out further below.
+ if (Stride->isZero())
+ return true;
+
+ auto *SE = PSE->getSE();
+ return SE->isKnownPositive(SE->getAbsExpr(Stride, false));
+ };
+ if (Seen.insert({Ptr, AccessTy}).second || !IsSafeReadWrite()) {
++NumReads;
IsReadOnlyPtr = true;
}
diff --git a/llvm/test/Analysis/LoopAccessAnalysis/single_strided_readwrite.ll b/llvm/test/Analysis/LoopAccessAnalysis/single_strided_readwrite.ll
index 390e694c0b340..43133aeb20531 100644
--- a/llvm/test/Analysis/LoopAccessAnalysis/single_strided_readwrite.ll
+++ b/llvm/test/Analysis/LoopAccessAnalysis/single_strided_readwrite.ll
@@ -4,13 +4,8 @@
define void @known_safe(ptr %p, i8 %a) {
; CHECK-LABEL: 'known_safe'
; CHECK-NEXT: header:
-; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
-; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Memory dependences are safe
; CHECK-NEXT: Dependences:
-; CHECK-NEXT: IndirectUnsafe:
-; CHECK-NEXT: %ld = load i64, ptr %gep, align 4 ->
-; CHECK-NEXT: store i64 %add, ptr %gep, align 4
-; CHECK-EMPTY:
; CHECK-NEXT: Run-time memory checks:
; CHECK-NEXT: Grouped accesses:
; CHECK-EMPTY:
@@ -44,13 +39,8 @@ exit:
define void @known_safe_byte_gep(ptr %p, i8 %a) {
; CHECK-LABEL: 'known_safe_byte_gep'
; CHECK-NEXT: header:
-; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
-; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Memory dependences are safe
; CHECK-NEXT: Dependences:
-; CHECK-NEXT: IndirectUnsafe:
-; CHECK-NEXT: %ld = load i64, ptr %gep, align 4 ->
-; CHECK-NEXT: store i64 %add, ptr %gep, align 4
-; CHECK-EMPTY:
; CHECK-NEXT: Run-time memory checks:
; CHECK-NEXT: Grouped accesses:
; CHECK-EMPTY:
@@ -241,3 +231,84 @@ header:
exit:
ret void
}
+
+; Not too important to actually support for now, the priority is to handle the
+; one below correctly.
+define void @known_safe_varying_stride(ptr %p, i8 %a) {
+; CHECK-LABEL: 'known_safe_varying_stride'
+; CHECK-NEXT: header:
+; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
+; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: IndirectUnsafe:
+; CHECK-NEXT: %ld = load i64, ptr %gep, align 4 ->
+; CHECK-NEXT: store i64 %add, ptr %gep, align 4
+; CHECK-EMPTY:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Grouped accesses:
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+entry:
+ br label %header
+
+header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %header ]
+ %iv.next = add nsw i64 %iv, 1
+ %mul = mul nsw nuw i64 %iv.next, %iv.next
+
+ %gep = getelementptr inbounds i64, ptr %p, i64 %mul
+ %ld = load i64, ptr %gep
+ %add = add i64 %ld, %iv
+ store i64 %add, ptr %gep
+
+ %exitcond = icmp slt i64 %iv.next, 128
+ br i1 %exitcond, label %header, label %exit
+
+exit:
+ ret void
+}
+
+define void @unsafe_varying_stride(ptr %p) {
+; CHECK-LABEL: 'unsafe_varying_stride'
+; CHECK-NEXT: header:
+; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
+; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: IndirectUnsafe:
+; CHECK-NEXT: %ld = load i64, ptr %gep, align 4 ->
+; CHECK-NEXT: store i64 %add, ptr %gep, align 4
+; CHECK-EMPTY:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Grouped accesses:
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+entry:
+ br label %header
+
+header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %header ]
+ %iv.next = add nsw i64 %iv, 1
+ %mul = mul nsw nuw i64 %iv.next, %iv.next
+
+ ; 0, 0, 3, ...
+ %idx = sub nsw nuw i64 %mul, %iv
+
+ %gep = getelementptr inbounds i64, ptr %p, i64 %idx
+ %ld = load i64, ptr %gep
+ %add = add i64 %ld, %iv
+ store i64 %add, ptr %gep
+
+ %exitcond = icmp slt i64 %iv.next, 128
+ br i1 %exitcond, label %header, label %exit
+
+exit:
+ ret void
+}
>From c9b0af9295095860e610e6a7e5c35d1341bd401b Mon Sep 17 00:00:00 2001
From: Andrei Elovikov <andrei.elovikov at sifive.com>
Date: Thu, 26 Mar 2026 10:11:51 -0700
Subject: [PATCH 4/5] Use `BadStride` helper for non-affine AddRec early return
---
llvm/lib/Analysis/LoopAccessAnalysis.cpp | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/lib/Analysis/LoopAccessAnalysis.cpp b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
index 133072b369e89..0fe173f0098a1 100644
--- a/llvm/lib/Analysis/LoopAccessAnalysis.cpp
+++ b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
@@ -993,7 +993,7 @@ const SCEV *llvm::getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp,
// Check the step is loop invariant.
if (!AR->isAffine())
- return nullptr;
+ return BadStride("Step is varying");
const SCEV *Step = AR->getStepRecurrence(*PSE.getSE());
>From 3e17cdd95827e6e77ef9ea9acc58e99ac61b249e Mon Sep 17 00:00:00 2001
From: Andrei Elovikov <andrei.elovikov at sifive.com>
Date: Wed, 29 Jul 2026 15:45:20 -0700
Subject: [PATCH 5/5] Address code review
---
llvm/lib/Analysis/LoopAccessAnalysis.cpp | 5 ++---
1 file changed, 2 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Analysis/LoopAccessAnalysis.cpp b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
index 0e45ba567d524..b1ba895254832 100644
--- a/llvm/lib/Analysis/LoopAccessAnalysis.cpp
+++ b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
@@ -1705,7 +1705,7 @@ std::optional<int64_t> llvm::getPtrStride(
PSE, AccessTy, Ptr, Lp, DT, StridesMap, ShouldCheckWrap, Predicates);
if (!StrideScev)
return std::nullopt;
- const APInt *APStride = nullptr;
+ const APInt *APStride;
if (!match(StrideScev, m_scev_APInt(APStride)))
return std::nullopt;
@@ -2823,8 +2823,7 @@ bool LoopAccessInfo::analyzeLoop(AAResults *AA, const LoopInfo *LI,
if (Stride->isZero())
return true;
- auto *SE = PSE->getSE();
- return SE->isKnownPositive(SE->getAbsExpr(Stride, false));
+ return PSE->getSE()->isKnownNonZero(Stride);
};
if (Seen.insert({Ptr, AccessTy}).second || !IsSafeReadWrite()) {
++NumReads;
More information about the llvm-commits
mailing list