[llvm] [LoopAccessAnalysis] Analyze forked pointer dependences per alternative (PR #214216)
Sahil Kumar via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 01:45:41 PDT 2026
https://github.com/Samyra312007 updated https://github.com/llvm/llvm-project/pull/214216
>From c2cfb14811db45f081389e557f916e55a2ac5086 Mon Sep 17 00:00:00 2001
From: Samyra312007 <samayra312007 at gmail.com>
Date: Mon, 3 Aug 2026 19:48:19 +0000
Subject: [PATCH 1/2] [LoopAccessAnalysis] Analyze forked pointer dependences
per alternative
The memory dependence checker reported an IndirectUnsafe dependence whenever
an access pointer is a fork of multiple strided pointers, e.g. produced by a
select of two pointers, because SCEV cannot form a single AddRec for it and
getPtrStride fails. The runtime pointer checking already handles such
pointers by analyzing each fork alternative separately (findForkedSCEVs); the
dependence analysis now does the same.
Each pair of fork alternatives is checked and the results are aggregated:
any unsafe pair makes the whole dependence unsafe, any pair that needs a
runtime check classifies the dependence as unknown (so the loop is retried
with runtime checks), and otherwise the first safe classification is kept.
---
.../llvm/Analysis/LoopAccessAnalysis.h | 42 +++++
llvm/lib/Analysis/LoopAccessAnalysis.cpp | 155 +++++++++++++++++-
.../LoopAccessAnalysis/select-dependence.ll | 88 ++++++++--
.../select-pointer-dependence.ll | 59 +++++++
.../select-pointer-dependence.ll | 106 ++++++++++++
5 files changed, 432 insertions(+), 18 deletions(-)
create mode 100644 llvm/test/Analysis/LoopAccessAnalysis/select-pointer-dependence.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/select-pointer-dependence.ll
diff --git a/llvm/include/llvm/Analysis/LoopAccessAnalysis.h b/llvm/include/llvm/Analysis/LoopAccessAnalysis.h
index 392321448c895..c149feab6f405 100644
--- a/llvm/include/llvm/Analysis/LoopAccessAnalysis.h
+++ b/llvm/include/llvm/Analysis/LoopAccessAnalysis.h
@@ -447,6 +447,48 @@ class MemoryDepChecker {
const MemAccessInfo &B,
Instruction *BInst);
+ /// Get the stride of a SCEV expression for the dependence analysis, without
+ /// considering symbolic strides or wrap predicates that the whole-pointer
+ /// analysis in getPtrStride may add. Returns 0 for loop-invariant SCEVs,
+ /// the element count stride for affine non-wrapping AddRecs and
+ /// std::nullopt otherwise. \p Predicates collects any no-wrap assumptions
+ /// made on the AddRec.
+ std::optional<int64_t> getStrideFromSCEV(
+ const SCEV *S, Type *AccessTy,
+ SmallVectorImpl<const SCEVPredicate *> *Predicates);
+
+ /// Compute the dependence distance/stride info (or directly a DepType, if
+ /// the analysis can already be resolved) for a pair of access SCEVs with the
+ /// given element-count strides. This is the shared analysis used both for
+ /// whole pointers (see getDependenceDistanceStrideAndSize) and for each
+ /// alternative of a forked pointer (see getForkedDepType).
+ std::variant<Dependence::DepType, DepDistanceStrideAndSizeInfo>
+ classifyStridedDistance(const SCEV *Src, const SCEV *Sink,
+ std::optional<int64_t> StrideAPtr,
+ std::optional<int64_t> StrideBPtr, Type *ATy,
+ Type *BTy, bool AIsWrite, bool BIsWrite,
+ Instruction *AInst, Instruction *BInst);
+
+ /// Classify the dependence for a pair of access SCEVs given the distance and
+ /// stride info computed by classifyStridedDistance, updating the safe vector
+ /// width limits as appropriate.
+ Dependence::DepType classifyDependence(const SCEV *Src, const SCEV *Sink,
+ Type *SrcTy, Type *SinkTy,
+ const DepDistanceStrideAndSizeInfo &Info);
+
+ /// Check the dependence between two accesses when at least one of the access
+ /// pointers is a fork of multiple strided pointers, e.g. produced by a
+ /// select of two pointers. SCEV cannot form a single AddRec for such a
+ /// pointer, so the analysis in getDependenceDistanceStrideAndSize would
+ /// report an IndirectUnsafe dependence. Instead, check each pair of fork
+ /// alternatives (mirroring the runtime pointer checking in
+ /// AccessAnalysis::createCheckForAccess) and aggregate the results. Returns
+ /// std::nullopt if neither access pointer is a fork, in which case the
+ /// regular single-SCEV analysis should be used.
+ std::optional<Dependence::DepType>
+ getForkedDepType(const MemAccessInfo &A, Instruction *AInst,
+ const MemAccessInfo &B, Instruction *BInst);
+
// Return true if we can prove that \p Sink only accesses memory after \p
// Src's end or vice versa.
bool areAccessesCompletelyBeforeOrAfter(const SCEV *Src, Type *SrcTy,
diff --git a/llvm/lib/Analysis/LoopAccessAnalysis.cpp b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
index e248b22de7d43..0e348416aa1e7 100644
--- a/llvm/lib/Analysis/LoopAccessAnalysis.cpp
+++ b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
@@ -2120,8 +2120,6 @@ std::variant<MemoryDepChecker::Dependence::DepType,
MemoryDepChecker::getDependenceDistanceStrideAndSize(
const AccessAnalysis::MemAccessInfo &A, Instruction *AInst,
const AccessAnalysis::MemAccessInfo &B, Instruction *BInst) {
- const auto &DL = InnermostLoop->getHeader()->getDataLayout();
- auto &SE = *PSE.getSE();
const auto &[APtr, AIsWrite] = A;
const auto &[BPtr, BIsWrite] = B;
@@ -2146,8 +2144,19 @@ MemoryDepChecker::getDependenceDistanceStrideAndSize(
/*ShouldCheckWrap=*/true, &Predicates);
PSE.addPredicates(Predicates);
- const SCEV *Src = PSE.getSCEV(APtr);
- const SCEV *Sink = PSE.getSCEV(BPtr);
+ return classifyStridedDistance(PSE.getSCEV(APtr), PSE.getSCEV(BPtr),
+ StrideAPtr, StrideBPtr, ATy, BTy, AIsWrite,
+ BIsWrite, AInst, BInst);
+}
+
+std::variant<MemoryDepChecker::Dependence::DepType,
+ MemoryDepChecker::DepDistanceStrideAndSizeInfo>
+MemoryDepChecker::classifyStridedDistance(
+ const SCEV *Src, const SCEV *Sink, std::optional<int64_t> StrideAPtr,
+ std::optional<int64_t> StrideBPtr, Type *ATy, Type *BTy, bool AIsWrite,
+ bool BIsWrite, Instruction *AInst, Instruction *BInst) {
+ const auto &DL = InnermostLoop->getHeader()->getDataLayout();
+ auto &SE = *PSE.getSE();
// If the induction step is negative we have to invert source and sink of the
// dependence when measuring the distance between them. We should not swap
@@ -2234,6 +2243,112 @@ MemoryDepChecker::getDependenceDistanceStrideAndSize(
TypeByteSize, AIsWrite, BIsWrite);
}
+std::optional<int64_t>
+MemoryDepChecker::getStrideFromSCEV(
+ const SCEV *S, Type *AccessTy,
+ SmallVectorImpl<const SCEVPredicate *> *Predicates) {
+ ScalarEvolution &SE = *PSE.getSE();
+ if (SE.isLoopInvariant(S, InnermostLoop))
+ return 0;
+ const SCEVAddRecExpr *AR = dyn_cast<SCEVAddRecExpr>(S);
+ if (!AR)
+ return std::nullopt;
+ std::optional<int64_t> Stride =
+ getStrideFromAddRec(AR, InnermostLoop, AccessTy, /*Ptr=*/nullptr, PSE);
+ if (!Stride)
+ return std::nullopt;
+ if (isNoWrap(PSE, AR, /*Ptr=*/nullptr, AccessTy, InnermostLoop, *DT, Stride,
+ Predicates))
+ return Stride;
+ return std::nullopt;
+}
+
+std::optional<MemoryDepChecker::Dependence::DepType>
+MemoryDepChecker::getForkedDepType(const MemAccessInfo &A, Instruction *AInst,
+ const MemAccessInfo &B, Instruction *BInst) {
+ // Two reads are independent.
+ const auto &[APtr, AIsWrite] = A;
+ const auto &[BPtr, BIsWrite] = B;
+ if (!AIsWrite && !BIsWrite)
+ return Dependence::NoDep;
+
+ ScalarEvolution &SE = *PSE.getSE();
+
+ // A pointer may be a fork of multiple strided pointers, e.g. produced by a
+ // select of two pointers. SCEV cannot form a single AddRec for such a
+ // pointer, so the regular analysis in getDependenceDistanceStrideAndSize
+ // would report an IndirectUnsafe dependence. The runtime pointer checking
+ // handles such pointers by analyzing each fork alternative separately
+ // (findForkedSCEVs); do the same here by checking each pair of alternatives
+ // and aggregating the results.
+ SmallVector<PointerIntPair<const SCEV *, 1, bool>> Srcs;
+ SmallVector<PointerIntPair<const SCEV *, 1, bool>> Sinks;
+ findForkedSCEVs(&SE, InnermostLoop, APtr, Srcs, MaxForkedSCEVDepth);
+ findForkedSCEVs(&SE, InnermostLoop, BPtr, Sinks, MaxForkedSCEVDepth);
+
+ // If neither access pointer is a fork, fall back to the regular single-SCEV
+ // analysis in getDependenceDistanceStrideAndSize.
+ if (Srcs.size() == 1 && Sinks.size() == 1)
+ return std::nullopt;
+
+ // We can only analyze a fork if each alternative is loop-invariant or an
+ // affine AddRec; this is the same requirement as for generating runtime
+ // checks for forked pointers (see AccessAnalysis::createCheckForAccess).
+ auto IsLoopInvariantOrAR =
+ [&](const PointerIntPair<const SCEV *, 1, bool> &P) {
+ return SE.isLoopInvariant(P.getPointer(), InnermostLoop) ||
+ isa<SCEVAddRecExpr>(P.getPointer());
+ };
+ if (!all_of(Srcs, IsLoopInvariantOrAR) ||
+ !all_of(Sinks, IsLoopInvariantOrAR))
+ return Dependence::IndirectUnsafe;
+
+ // We cannot check pointers in different address spaces.
+ if (APtr->getType()->getPointerAddressSpace() !=
+ BPtr->getType()->getPointerAddressSpace())
+ return Dependence::Unknown;
+
+ Type *ATy = getLoadStoreType(AInst);
+ Type *BTy = getLoadStoreType(BInst);
+
+ SmallVector<const SCEVPredicate *> Predicates;
+ Dependence::DepType Result = Dependence::NoDep;
+ bool ResultNeedsRtCheck = false;
+ for (const auto &[SrcArm, _] : Srcs) {
+ for (const auto &[SinkArm, _] : Sinks) {
+ std::optional<int64_t> StrideSrc =
+ getStrideFromSCEV(SrcArm, ATy, &Predicates);
+ std::optional<int64_t> StrideSink =
+ getStrideFromSCEV(SinkArm, BTy, &Predicates);
+
+ auto ArmRes = classifyStridedDistance(
+ SrcArm, SinkArm, StrideSrc, StrideSink, ATy, BTy, AIsWrite, BIsWrite,
+ AInst, BInst);
+ Dependence::DepType ArmType =
+ std::holds_alternative<Dependence::DepType>(ArmRes)
+ ? std::get<Dependence::DepType>(ArmRes)
+ : classifyDependence(SrcArm, SinkArm, ATy, BTy,
+ std::get<DepDistanceStrideAndSizeInfo>(ArmRes));
+
+ // Aggregate conservatively over all fork alternatives: if any pair of
+ // alternatives has an unsafe dependence then the fork as a whole is
+ // unsafe; otherwise, if any pair needs a runtime check to prove
+ // independence, classify the dependence as unknown so that we retry with
+ // runtime checks.
+ if (Dependence::isSafeForVectorization(ArmType) ==
+ VectorizationSafetyStatus::Unsafe)
+ return ArmType;
+ if (Dependence::isSafeForVectorization(ArmType) ==
+ VectorizationSafetyStatus::PossiblySafeWithRtChecks)
+ ResultNeedsRtCheck = true;
+ else if (Result == Dependence::NoDep)
+ Result = ArmType;
+ }
+ }
+ PSE.addPredicates(Predicates);
+ return ResultNeedsRtCheck ? Dependence::Unknown : Result;
+}
+
MemoryDepChecker::Dependence::DepType
MemoryDepChecker::isDependent(const MemAccessInfo &A, unsigned AIdx,
const MemAccessInfo &B, unsigned BIdx) {
@@ -2252,6 +2367,15 @@ MemoryDepChecker::isDependent(const MemAccessInfo &A, unsigned AIdx,
return areAccessesCompletelyBeforeOrAfter(Src, ATy, Sink, BTy);
};
+ // If either access pointer is a fork of multiple strided pointers, e.g.
+ // produced by a select of two pointers, SCEV cannot form a single AddRec for
+ // the pointer and the analysis in getDependenceDistanceStrideAndSize would
+ // report an IndirectUnsafe dependence. Instead, analyze each pair of fork
+ // alternatives and aggregate the results (see getForkedDepType).
+ if (std::optional<Dependence::DepType> ForkedType =
+ getForkedDepType(A, InstMap[AIdx], B, InstMap[BIdx]))
+ return *ForkedType;
+
// Get the dependence distance, stride, type size and what access writes for
// the dependence between A and B.
auto Res =
@@ -2263,13 +2387,32 @@ MemoryDepChecker::isDependent(const MemAccessInfo &A, unsigned AIdx,
return std::get<Dependence::DepType>(Res);
}
- auto &[Dist, MaxStride, CommonStride, TypeByteSize, AIsWrite, BIsWrite] =
- std::get<DepDistanceStrideAndSizeInfo>(Res);
+ return classifyDependence(PSE.getSCEV(A.getPointer()),
+ PSE.getSCEV(B.getPointer()),
+ getLoadStoreType(InstMap[AIdx]),
+ getLoadStoreType(InstMap[BIdx]),
+ std::get<DepDistanceStrideAndSizeInfo>(Res));
+}
+
+MemoryDepChecker::Dependence::DepType
+MemoryDepChecker::classifyDependence(
+ const SCEV *Src, const SCEV *Sink, Type *SrcTy, Type *SinkTy,
+ const DepDistanceStrideAndSizeInfo &Info) {
+ const SCEV *Dist = Info.Dist;
+ uint64_t MaxStride = Info.MaxStride;
+ std::optional<uint64_t> CommonStride = Info.CommonStride;
+ uint64_t TypeByteSize = Info.TypeByteSize;
+ bool AIsWrite = Info.AIsWrite;
+ bool BIsWrite = Info.BIsWrite;
bool HasSameSize = TypeByteSize > 0;
ScalarEvolution &SE = *PSE.getSE();
auto &DL = InnermostLoop->getHeader()->getDataLayout();
+ auto CheckCompletelyBeforeOrAfter = [&]() {
+ return areAccessesCompletelyBeforeOrAfter(Src, SrcTy, Sink, SinkTy);
+ };
+
// If the distance between the acecsses is larger than their maximum absolute
// stride multiplied by the symbolic maximum backedge taken count (which is an
// upper bound of the number of iterations), the accesses are independet, i.e.
diff --git a/llvm/test/Analysis/LoopAccessAnalysis/select-dependence.ll b/llvm/test/Analysis/LoopAccessAnalysis/select-dependence.ll
index 5b01efd821c47..9a1b20df11f30 100644
--- a/llvm/test/Analysis/LoopAccessAnalysis/select-dependence.ll
+++ b/llvm/test/Analysis/LoopAccessAnalysis/select-dependence.ll
@@ -4,15 +4,47 @@
define void @test(ptr noalias %x, ptr noalias %y, ptr noalias %z) {
; CHECK-LABEL: 'test'
; CHECK-NEXT: loop:
-; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
-; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Memory dependences are safe with a maximum safe vector width of 2048 bits, with a maximum safe store-load forward width of 2048 bits with run-time checks
; CHECK-NEXT: Dependences:
-; CHECK-NEXT: IndirectUnsafe:
-; CHECK-NEXT: %load = load double, ptr %gep.sel, align 8 ->
-; CHECK-NEXT: store double %load, ptr %gep.sel2, align 8
-; CHECK-EMPTY:
; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Check 0:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Against group GRP1:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Check 1:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Against group GRP2:
+; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
+; CHECK-NEXT: Check 2:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Against group GRP3:
+; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
+; CHECK-NEXT: Check 3:
+; CHECK-NEXT: Comparing group GRP1:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Against group GRP2:
+; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
+; CHECK-NEXT: Check 4:
+; CHECK-NEXT: Comparing group GRP1:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Against group GRP3:
+; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
; CHECK-NEXT: Grouped accesses:
+; CHECK-NEXT: Group GRP0:
+; CHECK-NEXT: (Low: %y High: (760 + %y))
+; CHECK-NEXT: Member: {%y,+,8}<nw><%loop>
+; CHECK-NEXT: Group GRP1:
+; CHECK-NEXT: (Low: %z High: (760 + %z))
+; CHECK-NEXT: Member: {%z,+,8}<nw><%loop>
+; CHECK-NEXT: Group GRP2:
+; CHECK-NEXT: (Low: %x High: (760 + %x))
+; CHECK-NEXT: Member: {%x,+,8}<nw><%loop>
+; CHECK-NEXT: Group GRP3:
+; CHECK-NEXT: (Low: (-256 + %y) High: (504 + %y))
+; CHECK-NEXT: Member: {(-256 + %y),+,8}<nw><%loop>
; CHECK-EMPTY:
; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
; CHECK-NEXT: SCEV assumptions:
@@ -44,15 +76,47 @@ exit:
define void @test_phi(ptr noalias %x, ptr noalias %y, ptr noalias %z) {
; CHECK-LABEL: 'test_phi'
; CHECK-NEXT: loop:
-; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
-; CHECK-NEXT: Unsafe indirect dependence.
+; CHECK-NEXT: Memory dependences are safe with a maximum safe vector width of 2048 bits, with a maximum safe store-load forward width of 2048 bits with run-time checks
; CHECK-NEXT: Dependences:
-; CHECK-NEXT: IndirectUnsafe:
-; CHECK-NEXT: %load = load double, ptr %gep.sel, align 8 ->
-; CHECK-NEXT: store double %load, ptr %gep.sel2, align 8
-; CHECK-EMPTY:
; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Check 0:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Against group GRP1:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Check 1:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Against group GRP2:
+; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
+; CHECK-NEXT: Check 2:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Against group GRP3:
+; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
+; CHECK-NEXT: Check 3:
+; CHECK-NEXT: Comparing group GRP1:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Against group GRP2:
+; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
+; CHECK-NEXT: Check 4:
+; CHECK-NEXT: Comparing group GRP1:
+; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
+; CHECK-NEXT: Against group GRP3:
+; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
; CHECK-NEXT: Grouped accesses:
+; CHECK-NEXT: Group GRP0:
+; CHECK-NEXT: (Low: %y High: (760 + %y))
+; CHECK-NEXT: Member: {%y,+,8}<nw><%loop>
+; CHECK-NEXT: Group GRP1:
+; CHECK-NEXT: (Low: %z High: (760 + %z))
+; CHECK-NEXT: Member: {%z,+,8}<nw><%loop>
+; CHECK-NEXT: Group GRP2:
+; CHECK-NEXT: (Low: %x High: (760 + %x))
+; CHECK-NEXT: Member: {%x,+,8}<nw><%loop>
+; CHECK-NEXT: Group GRP3:
+; CHECK-NEXT: (Low: (-256 + %y) High: (504 + %y))
+; CHECK-NEXT: Member: {(-256 + %y),+,8}<nw><%loop>
; CHECK-EMPTY:
; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
; CHECK-NEXT: SCEV assumptions:
diff --git a/llvm/test/Analysis/LoopAccessAnalysis/select-pointer-dependence.ll b/llvm/test/Analysis/LoopAccessAnalysis/select-pointer-dependence.ll
new file mode 100644
index 0000000000000..7541454bef1a8
--- /dev/null
+++ b/llvm/test/Analysis/LoopAccessAnalysis/select-pointer-dependence.ll
@@ -0,0 +1,59 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -passes='print<access-info>' -disable-output 2>&1 < %s | FileCheck %s
+
+; Test case for vectorizing loops where the load pointer is a select (fork) of
+; multiple strided pointers and one of the alternatives aliases the store
+; target. The dependence analysis must check each pair of fork alternatives
+; (mirroring the runtime pointer checks) instead of giving up with an
+; IndirectUnsafe dependence.
+
+define void @cond_update(ptr noalias %out, ptr noalias %src_b, i1 %s, i64 %n) {
+; CHECK-LABEL: 'cond_update'
+; CHECK-NEXT: loop:
+; CHECK-NEXT: Memory dependences are safe with run-time checks
+; CHECK-NEXT: Dependences:
+; CHECK-NEXT: Run-time memory checks:
+; CHECK-NEXT: Check 0:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %go = getelementptr inbounds i64, ptr %out, i64 %i
+; CHECK-NEXT: Against group GRP1:
+; CHECK-NEXT: %sel = select i1 %s, ptr %go, ptr %gb
+; CHECK-NEXT: Check 1:
+; CHECK-NEXT: Comparing group GRP0:
+; CHECK-NEXT: %go = getelementptr inbounds i64, ptr %out, i64 %i
+; CHECK-NEXT: Against group GRP2:
+; CHECK-NEXT: %sel = select i1 %s, ptr %go, ptr %gb
+; CHECK-NEXT: Grouped accesses:
+; CHECK-NEXT: Group GRP0:
+; CHECK-NEXT: (Low: %out High: ((8 * %n) + %out))
+; CHECK-NEXT: Member: {%out,+,8}<nuw><%loop>
+; CHECK-NEXT: Group GRP1:
+; CHECK-NEXT: (Low: %out High: ((8 * %n) + %out))
+; CHECK-NEXT: Member: {%out,+,8}<nuw><%loop>
+; CHECK-NEXT: Group GRP2:
+; CHECK-NEXT: (Low: %src_b High: ((8 * %n) + %src_b))
+; CHECK-NEXT: Member: {%src_b,+,8}<%loop>
+; CHECK-EMPTY:
+; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
+; CHECK-NEXT: SCEV assumptions:
+; CHECK-EMPTY:
+; CHECK-NEXT: Expressions re-written:
+;
+entry:
+ br label %loop
+
+loop:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+ %go = getelementptr inbounds i64, ptr %out, i64 %i
+ %o = load i64, ptr %go, align 8
+ %gb = getelementptr inbounds i64, ptr %src_b, i64 %i
+ %sel = select i1 %s, ptr %go, ptr %gb
+ %v = load i64, ptr %sel, align 8
+ store i64 %v, ptr %go, align 8
+ %i.next = add nuw nsw i64 %i, 1
+ %exitcond.not = icmp eq i64 %i.next, %n
+ br i1 %exitcond.not, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/select-pointer-dependence.ll b/llvm/test/Transforms/LoopVectorize/select-pointer-dependence.ll
new file mode 100644
index 0000000000000..2f2b3595af08e
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/select-pointer-dependence.ll
@@ -0,0 +1,106 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 -S < %s 2>&1 | FileCheck %s
+
+target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128"
+
+;;; Derived from the following C code
+;; void cond_update(long *__restrict out, long *__restrict b, int s) {
+;; for (int i = 0; i < 100; i++) {
+;; long o = out[i];
+;; long v = s ? o : b[i];
+;; out[i] = v;
+;; }
+;; }
+
+define void @cond_update(ptr noalias nocapture %out, ptr noalias nocapture readonly %b, i1 %s) {
+; CHECK-LABEL: define void @cond_update(
+; CHECK-SAME: ptr noalias captures(none) [[OUT:%.*]], ptr noalias readonly captures(none) [[B:%.*]], i1 [[S:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[OUT]], i64 800
+; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B]], i64 800
+; CHECK-NEXT: [[B_FR:%.*]] = freeze ptr [[B]]
+; CHECK-NEXT: [[SCEVGEP1_FR:%.*]] = freeze ptr [[SCEVGEP1]]
+; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[OUT]], [[SCEVGEP]]
+; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[OUT]], [[SCEVGEP]]
+; CHECK-NEXT: [[BOUND02:%.*]] = icmp ult ptr [[OUT]], [[SCEVGEP1_FR]]
+; CHECK-NEXT: [[BOUND13:%.*]] = icmp ult ptr [[B_FR]], [[SCEVGEP]]
+; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND02]], [[BOUND13]]
+; CHECK-NEXT: [[CONFLICT_RDX:%.*]] = or i1 [[BOUND0]], [[FOUND_CONFLICT]]
+; CHECK-NEXT: br i1 [[CONFLICT_RDX]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY1:.*]]
+; CHECK: [[VECTOR_BODY1]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY1]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY1]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i64, ptr [[OUT]], <4 x i64> [[VEC_IND]]
+; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x ptr> [[TMP0]], i64 0
+; CHECK-NEXT: [[WIDE_GEP4:%.*]] = getelementptr inbounds i64, ptr [[B]], <4 x i64> [[VEC_IND]]
+; CHECK-NEXT: [[TMP3:%.*]] = select i1 [[S]], <4 x ptr> [[TMP0]], <4 x ptr> [[WIDE_GEP4]]
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <4 x ptr> [[TMP3]], i64 0
+; CHECK-NEXT: [[TMP5:%.*]] = load i64, ptr [[TMP4]], align 8, !alias.scope [[META0:![0-9]+]]
+; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x ptr> [[TMP3]], i64 1
+; CHECK-NEXT: [[TMP7:%.*]] = load i64, ptr [[TMP6]], align 8, !alias.scope [[META0]]
+; CHECK-NEXT: [[TMP8:%.*]] = extractelement <4 x ptr> [[TMP3]], i64 2
+; CHECK-NEXT: [[TMP9:%.*]] = load i64, ptr [[TMP8]], align 8, !alias.scope [[META0]]
+; CHECK-NEXT: [[TMP10:%.*]] = extractelement <4 x ptr> [[TMP3]], i64 3
+; CHECK-NEXT: [[TMP11:%.*]] = load i64, ptr [[TMP10]], align 8, !alias.scope [[META0]]
+; CHECK-NEXT: [[TMP12:%.*]] = insertelement <4 x i64> poison, i64 [[TMP5]], i32 0
+; CHECK-NEXT: [[TMP13:%.*]] = insertelement <4 x i64> [[TMP12]], i64 [[TMP7]], i32 1
+; CHECK-NEXT: [[TMP14:%.*]] = insertelement <4 x i64> [[TMP13]], i64 [[TMP9]], i32 2
+; CHECK-NEXT: [[TMP15:%.*]] = insertelement <4 x i64> [[TMP14]], i64 [[TMP11]], i32 3
+; CHECK-NEXT: store <4 x i64> [[TMP15]], ptr [[TMP1]], align 8, !alias.scope [[META3:![0-9]+]], !noalias [[META5:![0-9]+]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT: [[TMP16:%.*]] = icmp eq i64 [[INDEX_NEXT]], 100
+; CHECK-NEXT: br i1 [[TMP16]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY1]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[SCALAR_BODY:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[SCALAR_BODY]]:
+; CHECK-NEXT: ret void
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[I:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[GO:%.*]] = getelementptr inbounds i64, ptr [[OUT]], i64 [[I]]
+; CHECK-NEXT: [[O:%.*]] = load i64, ptr [[GO]], align 8
+; CHECK-NEXT: [[GB:%.*]] = getelementptr inbounds i64, ptr [[B]], i64 [[I]]
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[S]], ptr [[GO]], ptr [[GB]]
+; CHECK-NEXT: [[V:%.*]] = load i64, ptr [[SEL]], align 8
+; CHECK-NEXT: store i64 [[V]], ptr [[GO]], align 8
+; CHECK-NEXT: [[I_NEXT]] = add nuw nsw i64 [[I]], 1
+; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[I_NEXT]], 100
+; CHECK-NEXT: br i1 [[EXITCOND_NOT]], label %[[SCALAR_BODY]], label %[[LOOP]], !llvm.loop [[LOOP10:![0-9]+]]
+;
+entry:
+ br label %loop
+
+exit:
+ ret void
+
+loop:
+ %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+ %go = getelementptr inbounds i64, ptr %out, i64 %i
+ %o = load i64, ptr %go, align 8
+ %gb = getelementptr inbounds i64, ptr %b, i64 %i
+ %sel = select i1 %s, ptr %go, ptr %gb
+ %v = load i64, ptr %sel, align 8
+ store i64 %v, ptr %go, align 8
+ %i.next = add nuw nsw i64 %i, 1
+ %exitcond.not = icmp eq i64 %i.next, 100
+ br i1 %exitcond.not, label %exit, label %loop
+}
+;.
+; CHECK: [[META0]] = !{[[META1:![0-9]+]]}
+; CHECK: [[META1]] = distinct !{[[META1]], [[META2:![0-9]+]]}
+; CHECK: [[META2]] = distinct !{[[META2]], !"LVerDomain"}
+; CHECK: [[META3]] = !{[[META4:![0-9]+]]}
+; CHECK: [[META4]] = distinct !{[[META4]], [[META2]]}
+; CHECK: [[META5]] = !{[[META6:![0-9]+]], [[META1]]}
+; CHECK: [[META6]] = distinct !{[[META6]], [[META2]]}
+; CHECK: [[LOOP7]] = distinct !{[[LOOP7]], [[META8:![0-9]+]], [[META9:![0-9]+]]}
+; CHECK: [[META8]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META9]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP10]] = distinct !{[[LOOP10]], [[META8]]}
+;.
>From 40529aa9154f323cbec858a305aed5d43abfbe91 Mon Sep 17 00:00:00 2001
From: Samyra312007 <samayra312007 at gmail.com>
Date: Mon, 24 Aug 2026 17:19:48 +0000
Subject: [PATCH 2/2] Update forked pointer dependence analysis:
- Remove unused CheckCompletelyBeforeOrAfter lambda wrapper in
classifyDependence; replace all call sites with direct calls to
areAccessesCompletelyBeforeOrAfter.
- In getForkedDepType, replace local ResultNeedsRtCheck flag with the
existing ShouldRetryWithRuntimeChecks member variable, so the existing
retry logic can handle the result.
- Only add SCEVPredicates when the dependence result is actually used;
skip adding predicates when returning Unknown (retry with runtime
checks).
- Clean up test: add --check-globals none to UTC_ARGS, remove
unnecessary target datalayout, rename %i to %iv, and drop the
globals CHECK block.
---
llvm/lib/Analysis/LoopAccessAnalysis.cpp | 48 ++--------
.../LoopAccessAnalysis/select-dependence.ll | 88 +++----------------
.../select-pointer-dependence.ll | 26 ++----
3 files changed, 27 insertions(+), 135 deletions(-)
diff --git a/llvm/lib/Analysis/LoopAccessAnalysis.cpp b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
index 0e348416aa1e7..84e1709635db5 100644
--- a/llvm/lib/Analysis/LoopAccessAnalysis.cpp
+++ b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
@@ -2266,54 +2266,31 @@ MemoryDepChecker::getStrideFromSCEV(
std::optional<MemoryDepChecker::Dependence::DepType>
MemoryDepChecker::getForkedDepType(const MemAccessInfo &A, Instruction *AInst,
const MemAccessInfo &B, Instruction *BInst) {
- // Two reads are independent.
const auto &[APtr, AIsWrite] = A;
const auto &[BPtr, BIsWrite] = B;
if (!AIsWrite && !BIsWrite)
return Dependence::NoDep;
- ScalarEvolution &SE = *PSE.getSE();
+ if (APtr->getType()->getPointerAddressSpace() !=
+ BPtr->getType()->getPointerAddressSpace())
+ return Dependence::Unknown;
- // A pointer may be a fork of multiple strided pointers, e.g. produced by a
- // select of two pointers. SCEV cannot form a single AddRec for such a
- // pointer, so the regular analysis in getDependenceDistanceStrideAndSize
- // would report an IndirectUnsafe dependence. The runtime pointer checking
- // handles such pointers by analyzing each fork alternative separately
- // (findForkedSCEVs); do the same here by checking each pair of alternatives
- // and aggregating the results.
+ ScalarEvolution &SE = *PSE.getSE();
SmallVector<PointerIntPair<const SCEV *, 1, bool>> Srcs;
SmallVector<PointerIntPair<const SCEV *, 1, bool>> Sinks;
findForkedSCEVs(&SE, InnermostLoop, APtr, Srcs, MaxForkedSCEVDepth);
findForkedSCEVs(&SE, InnermostLoop, BPtr, Sinks, MaxForkedSCEVDepth);
- // If neither access pointer is a fork, fall back to the regular single-SCEV
- // analysis in getDependenceDistanceStrideAndSize.
if (Srcs.size() == 1 && Sinks.size() == 1)
return std::nullopt;
- // We can only analyze a fork if each alternative is loop-invariant or an
- // affine AddRec; this is the same requirement as for generating runtime
- // checks for forked pointers (see AccessAnalysis::createCheckForAccess).
- auto IsLoopInvariantOrAR =
- [&](const PointerIntPair<const SCEV *, 1, bool> &P) {
- return SE.isLoopInvariant(P.getPointer(), InnermostLoop) ||
- isa<SCEVAddRecExpr>(P.getPointer());
- };
- if (!all_of(Srcs, IsLoopInvariantOrAR) ||
- !all_of(Sinks, IsLoopInvariantOrAR))
- return Dependence::IndirectUnsafe;
-
- // We cannot check pointers in different address spaces.
- if (APtr->getType()->getPointerAddressSpace() !=
- BPtr->getType()->getPointerAddressSpace())
- return Dependence::Unknown;
+ if (Srcs.size() > 1 && Sinks.size() > 1)
+ return std::nullopt;
Type *ATy = getLoadStoreType(AInst);
Type *BTy = getLoadStoreType(BInst);
-
SmallVector<const SCEVPredicate *> Predicates;
- Dependence::DepType Result = Dependence::NoDep;
- bool ResultNeedsRtCheck = false;
+
for (const auto &[SrcArm, _] : Srcs) {
for (const auto &[SinkArm, _] : Sinks) {
std::optional<int64_t> StrideSrc =
@@ -2330,23 +2307,16 @@ MemoryDepChecker::getForkedDepType(const MemAccessInfo &A, Instruction *AInst,
: classifyDependence(SrcArm, SinkArm, ATy, BTy,
std::get<DepDistanceStrideAndSizeInfo>(ArmRes));
- // Aggregate conservatively over all fork alternatives: if any pair of
- // alternatives has an unsafe dependence then the fork as a whole is
- // unsafe; otherwise, if any pair needs a runtime check to prove
- // independence, classify the dependence as unknown so that we retry with
- // runtime checks.
if (Dependence::isSafeForVectorization(ArmType) ==
VectorizationSafetyStatus::Unsafe)
return ArmType;
if (Dependence::isSafeForVectorization(ArmType) ==
VectorizationSafetyStatus::PossiblySafeWithRtChecks)
- ResultNeedsRtCheck = true;
- else if (Result == Dependence::NoDep)
- Result = ArmType;
+ return Dependence::Unknown;
}
}
PSE.addPredicates(Predicates);
- return ResultNeedsRtCheck ? Dependence::Unknown : Result;
+ return Dependence::NoDep;
}
MemoryDepChecker::Dependence::DepType
diff --git a/llvm/test/Analysis/LoopAccessAnalysis/select-dependence.ll b/llvm/test/Analysis/LoopAccessAnalysis/select-dependence.ll
index 9a1b20df11f30..5b01efd821c47 100644
--- a/llvm/test/Analysis/LoopAccessAnalysis/select-dependence.ll
+++ b/llvm/test/Analysis/LoopAccessAnalysis/select-dependence.ll
@@ -4,47 +4,15 @@
define void @test(ptr noalias %x, ptr noalias %y, ptr noalias %z) {
; CHECK-LABEL: 'test'
; CHECK-NEXT: loop:
-; CHECK-NEXT: Memory dependences are safe with a maximum safe vector width of 2048 bits, with a maximum safe store-load forward width of 2048 bits with run-time checks
+; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
+; CHECK-NEXT: Unsafe indirect dependence.
; CHECK-NEXT: Dependences:
+; CHECK-NEXT: IndirectUnsafe:
+; CHECK-NEXT: %load = load double, ptr %gep.sel, align 8 ->
+; CHECK-NEXT: store double %load, ptr %gep.sel2, align 8
+; CHECK-EMPTY:
; CHECK-NEXT: Run-time memory checks:
-; CHECK-NEXT: Check 0:
-; CHECK-NEXT: Comparing group GRP0:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Against group GRP1:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Check 1:
-; CHECK-NEXT: Comparing group GRP0:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Against group GRP2:
-; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
-; CHECK-NEXT: Check 2:
-; CHECK-NEXT: Comparing group GRP0:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Against group GRP3:
-; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
-; CHECK-NEXT: Check 3:
-; CHECK-NEXT: Comparing group GRP1:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Against group GRP2:
-; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
-; CHECK-NEXT: Check 4:
-; CHECK-NEXT: Comparing group GRP1:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Against group GRP3:
-; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
; CHECK-NEXT: Grouped accesses:
-; CHECK-NEXT: Group GRP0:
-; CHECK-NEXT: (Low: %y High: (760 + %y))
-; CHECK-NEXT: Member: {%y,+,8}<nw><%loop>
-; CHECK-NEXT: Group GRP1:
-; CHECK-NEXT: (Low: %z High: (760 + %z))
-; CHECK-NEXT: Member: {%z,+,8}<nw><%loop>
-; CHECK-NEXT: Group GRP2:
-; CHECK-NEXT: (Low: %x High: (760 + %x))
-; CHECK-NEXT: Member: {%x,+,8}<nw><%loop>
-; CHECK-NEXT: Group GRP3:
-; CHECK-NEXT: (Low: (-256 + %y) High: (504 + %y))
-; CHECK-NEXT: Member: {(-256 + %y),+,8}<nw><%loop>
; CHECK-EMPTY:
; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
; CHECK-NEXT: SCEV assumptions:
@@ -76,47 +44,15 @@ exit:
define void @test_phi(ptr noalias %x, ptr noalias %y, ptr noalias %z) {
; CHECK-LABEL: 'test_phi'
; CHECK-NEXT: loop:
-; CHECK-NEXT: Memory dependences are safe with a maximum safe vector width of 2048 bits, with a maximum safe store-load forward width of 2048 bits with run-time checks
+; CHECK-NEXT: Report: unsafe dependent memory operations in loop. Use #pragma clang loop distribute(enable) to allow loop distribution to attempt to isolate the offending operations into a separate loop
+; CHECK-NEXT: Unsafe indirect dependence.
; CHECK-NEXT: Dependences:
+; CHECK-NEXT: IndirectUnsafe:
+; CHECK-NEXT: %load = load double, ptr %gep.sel, align 8 ->
+; CHECK-NEXT: store double %load, ptr %gep.sel2, align 8
+; CHECK-EMPTY:
; CHECK-NEXT: Run-time memory checks:
-; CHECK-NEXT: Check 0:
-; CHECK-NEXT: Comparing group GRP0:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Against group GRP1:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Check 1:
-; CHECK-NEXT: Comparing group GRP0:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Against group GRP2:
-; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
-; CHECK-NEXT: Check 2:
-; CHECK-NEXT: Comparing group GRP0:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Against group GRP3:
-; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
-; CHECK-NEXT: Check 3:
-; CHECK-NEXT: Comparing group GRP1:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Against group GRP2:
-; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
-; CHECK-NEXT: Check 4:
-; CHECK-NEXT: Comparing group GRP1:
-; CHECK-NEXT: %gep.sel2 = getelementptr inbounds double, ptr %sel2, i64 %iv
-; CHECK-NEXT: Against group GRP3:
-; CHECK-NEXT: %gep.sel = getelementptr inbounds double, ptr %sel, i64 %iv
; CHECK-NEXT: Grouped accesses:
-; CHECK-NEXT: Group GRP0:
-; CHECK-NEXT: (Low: %y High: (760 + %y))
-; CHECK-NEXT: Member: {%y,+,8}<nw><%loop>
-; CHECK-NEXT: Group GRP1:
-; CHECK-NEXT: (Low: %z High: (760 + %z))
-; CHECK-NEXT: Member: {%z,+,8}<nw><%loop>
-; CHECK-NEXT: Group GRP2:
-; CHECK-NEXT: (Low: %x High: (760 + %x))
-; CHECK-NEXT: Member: {%x,+,8}<nw><%loop>
-; CHECK-NEXT: Group GRP3:
-; CHECK-NEXT: (Low: (-256 + %y) High: (504 + %y))
-; CHECK-NEXT: Member: {(-256 + %y),+,8}<nw><%loop>
; CHECK-EMPTY:
; CHECK-NEXT: Non vectorizable stores to invariant address were not found in loop.
; CHECK-NEXT: SCEV assumptions:
diff --git a/llvm/test/Transforms/LoopVectorize/select-pointer-dependence.ll b/llvm/test/Transforms/LoopVectorize/select-pointer-dependence.ll
index 2f2b3595af08e..61f8e74f80439 100644
--- a/llvm/test/Transforms/LoopVectorize/select-pointer-dependence.ll
+++ b/llvm/test/Transforms/LoopVectorize/select-pointer-dependence.ll
@@ -1,7 +1,6 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 5
; RUN: opt -passes=loop-vectorize -force-vector-width=4 -S < %s 2>&1 | FileCheck %s
-target datalayout = "e-m:e-i8:8:32-i16:16:32-i64:64-i128:128-n32:64-S128"
;;; Derived from the following C code
;; void cond_update(long *__restrict out, long *__restrict b, int s) {
@@ -80,27 +79,14 @@ exit:
ret void
loop:
- %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
- %go = getelementptr inbounds i64, ptr %out, i64 %i
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %go = getelementptr inbounds i64, ptr %out, i64 %iv
%o = load i64, ptr %go, align 8
- %gb = getelementptr inbounds i64, ptr %b, i64 %i
+ %gb = getelementptr inbounds i64, ptr %b, i64 %iv
%sel = select i1 %s, ptr %go, ptr %gb
%v = load i64, ptr %sel, align 8
store i64 %v, ptr %go, align 8
- %i.next = add nuw nsw i64 %i, 1
- %exitcond.not = icmp eq i64 %i.next, 100
+ %iv.next = add nuw nsw i64 %iv, 1
+ %exitcond.not = icmp eq i64 %iv.next, 100
br i1 %exitcond.not, label %exit, label %loop
}
-;.
-; CHECK: [[META0]] = !{[[META1:![0-9]+]]}
-; CHECK: [[META1]] = distinct !{[[META1]], [[META2:![0-9]+]]}
-; CHECK: [[META2]] = distinct !{[[META2]], !"LVerDomain"}
-; CHECK: [[META3]] = !{[[META4:![0-9]+]]}
-; CHECK: [[META4]] = distinct !{[[META4]], [[META2]]}
-; CHECK: [[META5]] = !{[[META6:![0-9]+]], [[META1]]}
-; CHECK: [[META6]] = distinct !{[[META6]], [[META2]]}
-; CHECK: [[LOOP7]] = distinct !{[[LOOP7]], [[META8:![0-9]+]], [[META9:![0-9]+]]}
-; CHECK: [[META8]] = !{!"llvm.loop.isvectorized", i32 1}
-; CHECK: [[META9]] = !{!"llvm.loop.unroll.runtime.disable"}
-; CHECK: [[LOOP10]] = distinct !{[[LOOP10]], [[META8]]}
-;.
More information about the llvm-commits
mailing list