[llvm] [SLP]Support runtime alias checks versioning for blocks in loops (PR #226955)
Alexey Bataev via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 29 04:03:51 PDT 2026
https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/226955
>From 030bad4e568ba6b95724ff7aa66656e1a3a6c088 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Mon, 28 Sep 2026 03:52:50 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
=?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Created using spr 1.3.7
---
.../Transforms/Vectorize/SLPVectorizer.cpp | 75 +++++++++++--------
.../SLPVectorizer/X86/runtime-alias-checks.ll | 44 +++++++++--
2 files changed, 78 insertions(+), 41 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 442cc9a0895b2..e2e9f987933c0 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -541,10 +541,10 @@ class slpvectorizer::BoUpSLP {
void captureRuntimeCheckBodySnapshot();
/// Returns true if \p BB satisfies the block-level preconditions for runtime
- /// alias check versioning (straight-line, outside any loop, duplicable, not a
- /// scalar fallback, function not optimized for size). These checks do not
- /// depend on the collected checks, so they can gate the (expensive)
- /// optimistic retry before any tree is rebuilt.
+ /// alias check versioning (straight-line, duplicable, not a scalar fallback,
+ /// function not optimized for size). These checks do not depend on the
+ /// collected checks, so they can gate the (expensive) optimistic retry before
+ /// any tree is rebuilt.
bool canVersionBlockForRuntimeChecks(BasicBlock *BB) const;
/// Returns true if the runtime alias checks can be safely emitted to guard
@@ -25363,10 +25363,6 @@ bool BoUpSLP::canVersionBlockForRuntimeChecks(BasicBlock *BB) const {
return false;
if (!DT->isReachableFromEntry(BB))
return false;
- // Versioning duplicates the block body; only straight-line code outside any
- // loop is handled for now, to avoid LoopInfo and region updates.
- if (LI->getLoopFor(BB))
- return false;
// Never version a scalar fallback block: it is the safe, original-order copy
// taken when aliasing is detected and must stay scalar.
if (ScalarFallbackBlocks.contains(BB))
@@ -25425,10 +25421,11 @@ bool BoUpSLP::canVersionForRuntimeChecks() {
Bases.insert(P.second);
}
// Every base object must be available in the (PHI-only) header where the
- // guard branch is emitted.
+ // guard branch is emitted, i.e. be one of its PHIs or dominate it.
if (any_of(Bases, [&](const Value *Base) {
const auto *I = dyn_cast<Instruction>(Base);
- return I && (I->getParent() == BB || !DT->dominates(I, BB));
+ return I && (I->getParent() == BB ? !isa<PHINode>(I)
+ : !DT->dominates(I, BB));
}))
return false;
@@ -25486,40 +25483,32 @@ bool BoUpSLP::canVersionForRuntimeChecks() {
It->second.second = SE->getUMaxExpr(It->second.second, EndOff);
}
}
+ // Every involved base must contribute at least one bounded access.
+ if (OffBounds.size() != Bases.size())
+ return false;
// Materialize the absolute [Low, High) = base + [minOff, maxEndOff). The
// offsets are constants, so each bound is a base-plus-constant expression
// that expands to a single add off the base address (no runtime umin/umax).
+ // The base is used as an opaque value, so the bounds only reference the base
+ // itself, available at the guard, and a loop-variant base (e.g. a header PHI)
+ // is reused rather than recomputed from its recurrence.
for (const auto &[Base, Off] : OffBounds) {
- const SCEV *BaseSC = SE->getPtrToAddrExpr(SE->getSCEV(Base));
+ const SCEV *BaseSC = SE->getPtrToAddrExpr(SE->getUnknown(Base));
RTChecks.Bounds.try_emplace(Base, SE->getAddExpr(BaseSC, Off.first),
SE->getAddExpr(BaseSC, Off.second));
}
- // Every involved base must contribute at least one bounded access, and its
- // bounds must be expandable at the guard (so the checks only reference values
- // available in the header).
- SCEVExpander Exp(*SE, "slp.rtcheck");
- // The guard is emitted after the header PHIs, so validate expandability at
- // the first non-PHI: only values available in the header (PHIs, arguments,
- // values defined before the block) dominate that point.
- Instruction *GuardPt = &*BB->getFirstNonPHIIt();
- if (any_of(Bases, [&](const Value *Base) {
- auto *It = RTChecks.Bounds.find(Base);
- return It == RTChecks.Bounds.end() ||
- !Exp.isSafeToExpandAt(It->second.first, GuardPt) ||
- !Exp.isSafeToExpandAt(It->second.second, GuardPt);
- }))
- return false;
- // No SSA value defined in the body may escape the versioned region: that
- // would require a merge PHI in the continuation block, which is not yet
- // supported. A use by the terminator counts as an escape because the
- // terminator is moved into the continuation block.
+ // SSA values defined in the body and used outside of it are merged from both
+ // paths in the continuation block, so they must stay scalar. Uses by the
+ // terminator (moved into the continuation block) and by the block's PHIs
+ // (via the loop latch) count as outside.
for (Instruction &I : *BB) {
- if (isa<PHINode>(&I) || I.isTerminator())
+ if (isa<PHINode>(&I) || I.isTerminator() || !isVectorized(&I))
continue;
if (any_of(I.users(), [&](User *U) {
auto *UI = dyn_cast<Instruction>(U);
- return !UI || UI->getParent() != BB || UI->isTerminator();
+ return !UI || UI->getParent() != BB || isa<PHINode>(UI) ||
+ UI->isTerminator();
}))
return false;
}
@@ -25682,6 +25671,21 @@ void BoUpSLP::versionBlocksForRuntimeChecks() {
for (BasicBlock *Succ : successors(Term))
Succ->replacePhiUsesWith(BB, Tail);
+ // Merge the body values used outside of the versioned region (by the
+ // terminator, across the loop latch or in other blocks) from both paths. The
+ // SCEVs cached for the outside users refer to the body values, which no
+ // longer dominate them.
+ for (Instruction *I : RTOrigBodyOrder) {
+ if (!I->isUsedOutsideOfBlock(VecBB))
+ continue;
+ SE->forgetValue(I);
+ PHINode *Merge = PHINode::Create(I->getType(), 2, I->getName() + ".rtmerge",
+ Tail->getFirstNonPHIIt());
+ I->replaceUsesOutsideBlock(Merge, VecBB);
+ Merge->addIncoming(I, VecBB);
+ Merge->addIncoming(VMap.lookup(I), ScalarBB);
+ }
+
// Keep the dominator tree valid for the remainder of the run.
if (DT) {
DomTreeUpdater DTU(*DT, DomTreeUpdater::UpdateStrategy::Eager);
@@ -25704,6 +25708,11 @@ void BoUpSLP::versionBlocksForRuntimeChecks() {
// instructions across blocks for nodes emitted after versioning.
DT->updateDFSNumbers();
}
+ // Keep the loop info valid for the remainder of the run: the new blocks
+ // belong to the loop of the versioned block.
+ if (Loop *L = LI->getLoopFor(BB))
+ for (BasicBlock *NewBB : {VecBB, ScalarBB, Tail})
+ L->addBasicBlockToLoop(NewBB, *LI);
// Record the checked base pairs for the fast-path block so that subsequent
// vectorization there can reuse this guard instead of versioning again.
@@ -29160,7 +29169,7 @@ SLPVectorizerPass::vectorizeStoreChain(ArrayRef<Value *> Chain, BoUpSLP &R,
if (R.runtimeChecksFailedForBlock(BB))
return Res;
// The retry rebuilds the whole tree; also skip it when the chain's block can
- // never be versioned (e.g. it is inside a loop).
+ // never be versioned (e.g. the function is optimized for size).
if (!R.canVersionBlockForRuntimeChecks(BB)) {
R.markRuntimeChecksFailedForBlock(BB);
return Res;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll b/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll
index 92db6896d8f42..9af4179a24940 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll
@@ -1630,9 +1630,33 @@ define void @versioned_loop_block(ptr %dst, i64 %stride, ptr %src, i64 %n) {
; CHECK-NEXT: [[LOOP_RTVEC:.*]]:
; CHECK-NEXT: br label %[[LOOP_RTCONT:.*]]
; CHECK: [[LOOP_RTCONT]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[LOOP_RTVEC]] ], [ [[IV_NEXT_SCALAR:%.*]], %[[LOOP_RTCONT]] ]
-; CHECK-NEXT: [[D:%.*]] = phi ptr [ [[DST]], %[[LOOP_RTVEC]] ], [ [[D_NEXT_SCALAR:%.*]], %[[LOOP_RTCONT]] ]
-; CHECK-NEXT: [[S:%.*]] = phi ptr [ [[SRC]], %[[LOOP_RTVEC]] ], [ [[S_NEXT_SCALAR:%.*]], %[[LOOP_RTCONT]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[LOOP_RTVEC]] ], [ [[IV_NEXT_RTMERGE:%.*]], %[[EXIT:.*]] ]
+; CHECK-NEXT: [[D:%.*]] = phi ptr [ [[DST]], %[[LOOP_RTVEC]] ], [ [[D_NEXT_RTMERGE:%.*]], %[[EXIT]] ]
+; CHECK-NEXT: [[S:%.*]] = phi ptr [ [[SRC]], %[[LOOP_RTVEC]] ], [ [[S_NEXT_RTMERGE:%.*]], %[[EXIT]] ]
+; CHECK-NEXT: [[S8:%.*]] = ptrtoaddr ptr [[S]] to i64
+; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[S8]], 16
+; CHECK-NEXT: [[D9:%.*]] = ptrtoaddr ptr [[D]] to i64
+; CHECK-NEXT: [[TMP1:%.*]] = add i64 [[D9]], 8
+; CHECK-NEXT: [[RT_BOUND0:%.*]] = icmp ult i64 [[D9]], [[TMP0]]
+; CHECK-NEXT: [[RT_BOUND1:%.*]] = icmp ult i64 [[S8]], [[TMP1]]
+; CHECK-NEXT: [[RT_CONFLICT:%.*]] = and i1 [[RT_BOUND0]], [[RT_BOUND1]]
+; CHECK-NEXT: [[RT_GUARD:%.*]] = freeze i1 [[RT_CONFLICT]]
+; CHECK-NEXT: br i1 [[RT_GUARD]], label %[[LOOP_RTSCALAR:.*]], label %[[LOOP_RTVEC1:.*]], !prof [[PROF0]]
+; CHECK: [[EXIT1:.*]]:
+; CHECK-NEXT: ret void
+; CHECK: [[LOOP_RTVEC1]]:
+; CHECK-NEXT: [[TMP2:%.*]] = load <8 x i16>, ptr [[S]], align 2
+; CHECK-NEXT: [[TMP3:%.*]] = ashr <8 x i16> [[TMP2]], splat (i16 8)
+; CHECK-NEXT: [[TMP4:%.*]] = sub <8 x i16> [[TMP2]], [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = lshr <8 x i16> [[TMP4]], splat (i16 6)
+; CHECK-NEXT: [[TMP6:%.*]] = trunc <8 x i16> [[TMP5]] to <8 x i8>
+; CHECK-NEXT: store <8 x i8> [[TMP6]], ptr [[D]], align 1
+; CHECK-NEXT: [[D_NEXT:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[STRIDE]]
+; CHECK-NEXT: [[S_NEXT:%.*]] = getelementptr inbounds nuw i8, ptr [[S]], i64 16
+; CHECK-NEXT: [[IV_NEXT:%.*]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT: [[COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[LOOP_RTSCALAR]]:
; CHECK-NEXT: [[L0_SCALAR:%.*]] = load i16, ptr [[S]], align 2
; CHECK-NEXT: [[A0_SCALAR:%.*]] = ashr i16 [[L0_SCALAR]], 8
; CHECK-NEXT: [[B0_SCALAR:%.*]] = sub i16 [[L0_SCALAR]], [[A0_SCALAR]]
@@ -1695,13 +1719,17 @@ define void @versioned_loop_block(ptr %dst, i64 %stride, ptr %src, i64 %n) {
; CHECK-NEXT: [[C7_SCALAR:%.*]] = lshr i16 [[B7_SCALAR]], 6
; CHECK-NEXT: [[T7_SCALAR:%.*]] = trunc i16 [[C7_SCALAR]] to i8
; CHECK-NEXT: store i8 [[T7_SCALAR]], ptr [[D7_SCALAR]], align 1
-; CHECK-NEXT: [[D_NEXT_SCALAR]] = getelementptr inbounds i8, ptr [[D]], i64 [[STRIDE]]
-; CHECK-NEXT: [[S_NEXT_SCALAR]] = getelementptr inbounds nuw i8, ptr [[S]], i64 16
-; CHECK-NEXT: [[IV_NEXT_SCALAR]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT: [[D_NEXT_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[STRIDE]]
+; CHECK-NEXT: [[S_NEXT_SCALAR:%.*]] = getelementptr inbounds nuw i8, ptr [[S]], i64 16
+; CHECK-NEXT: [[IV_NEXT_SCALAR:%.*]] = add nuw i64 [[IV]], 1
; CHECK-NEXT: [[COND_SCALAR:%.*]] = icmp eq i64 [[IV_NEXT_SCALAR]], [[N]]
-; CHECK-NEXT: br i1 [[COND_SCALAR]], label %[[EXIT:.*]], label %[[LOOP_RTCONT]]
+; CHECK-NEXT: br label %[[EXIT]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: ret void
+; CHECK-NEXT: [[D_NEXT_RTMERGE]] = phi ptr [ [[D_NEXT]], %[[LOOP_RTVEC1]] ], [ [[D_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT: [[S_NEXT_RTMERGE]] = phi ptr [ [[S_NEXT]], %[[LOOP_RTVEC1]] ], [ [[S_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT: [[IV_NEXT_RTMERGE]] = phi i64 [ [[IV_NEXT]], %[[LOOP_RTVEC1]] ], [ [[IV_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT: [[COND_RTMERGE:%.*]] = phi i1 [ [[COND]], %[[LOOP_RTVEC1]] ], [ [[COND_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT: br i1 [[COND_RTMERGE]], label %[[EXIT1]], label %[[LOOP_RTCONT]]
;
; NOCHK-LABEL: define void @versioned_loop_block(
; NOCHK-SAME: ptr [[DST:%.*]], i64 [[STRIDE:%.*]], ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR1]] {
More information about the llvm-commits
mailing list