[llvm] [SLP]Support runtime alias checks versioning for blocks in loops (PR #226955)

Alexey Bataev via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 29 04:03:51 PDT 2026


https://github.com/alexey-bataev updated https://github.com/llvm/llvm-project/pull/226955

>From 030bad4e568ba6b95724ff7aa66656e1a3a6c088 Mon Sep 17 00:00:00 2001
From: Alexey Bataev <a.bataev at outlook.com>
Date: Mon, 28 Sep 2026 03:52:50 -0700
Subject: [PATCH] =?UTF-8?q?[=F0=9D=98=80=F0=9D=97=BD=F0=9D=97=BF]=20initia?=
 =?UTF-8?q?l=20version?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit

Created using spr 1.3.7
---
 .../Transforms/Vectorize/SLPVectorizer.cpp    | 75 +++++++++++--------
 .../SLPVectorizer/X86/runtime-alias-checks.ll | 44 +++++++++--
 2 files changed, 78 insertions(+), 41 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 442cc9a0895b2..e2e9f987933c0 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -541,10 +541,10 @@ class slpvectorizer::BoUpSLP {
   void captureRuntimeCheckBodySnapshot();
 
   /// Returns true if \p BB satisfies the block-level preconditions for runtime
-  /// alias check versioning (straight-line, outside any loop, duplicable, not a
-  /// scalar fallback, function not optimized for size). These checks do not
-  /// depend on the collected checks, so they can gate the (expensive)
-  /// optimistic retry before any tree is rebuilt.
+  /// alias check versioning (straight-line, duplicable, not a scalar fallback,
+  /// function not optimized for size). These checks do not depend on the
+  /// collected checks, so they can gate the (expensive) optimistic retry before
+  /// any tree is rebuilt.
   bool canVersionBlockForRuntimeChecks(BasicBlock *BB) const;
 
   /// Returns true if the runtime alias checks can be safely emitted to guard
@@ -25363,10 +25363,6 @@ bool BoUpSLP::canVersionBlockForRuntimeChecks(BasicBlock *BB) const {
     return false;
   if (!DT->isReachableFromEntry(BB))
     return false;
-  // Versioning duplicates the block body; only straight-line code outside any
-  // loop is handled for now, to avoid LoopInfo and region updates.
-  if (LI->getLoopFor(BB))
-    return false;
   // Never version a scalar fallback block: it is the safe, original-order copy
   // taken when aliasing is detected and must stay scalar.
   if (ScalarFallbackBlocks.contains(BB))
@@ -25425,10 +25421,11 @@ bool BoUpSLP::canVersionForRuntimeChecks() {
     Bases.insert(P.second);
   }
   // Every base object must be available in the (PHI-only) header where the
-  // guard branch is emitted.
+  // guard branch is emitted, i.e. be one of its PHIs or dominate it.
   if (any_of(Bases, [&](const Value *Base) {
         const auto *I = dyn_cast<Instruction>(Base);
-        return I && (I->getParent() == BB || !DT->dominates(I, BB));
+        return I && (I->getParent() == BB ? !isa<PHINode>(I)
+                                          : !DT->dominates(I, BB));
       }))
     return false;
 
@@ -25486,40 +25483,32 @@ bool BoUpSLP::canVersionForRuntimeChecks() {
       It->second.second = SE->getUMaxExpr(It->second.second, EndOff);
     }
   }
+  // Every involved base must contribute at least one bounded access.
+  if (OffBounds.size() != Bases.size())
+    return false;
   // Materialize the absolute [Low, High) = base + [minOff, maxEndOff). The
   // offsets are constants, so each bound is a base-plus-constant expression
   // that expands to a single add off the base address (no runtime umin/umax).
+  // The base is used as an opaque value, so the bounds only reference the base
+  // itself, available at the guard, and a loop-variant base (e.g. a header PHI)
+  // is reused rather than recomputed from its recurrence.
   for (const auto &[Base, Off] : OffBounds) {
-    const SCEV *BaseSC = SE->getPtrToAddrExpr(SE->getSCEV(Base));
+    const SCEV *BaseSC = SE->getPtrToAddrExpr(SE->getUnknown(Base));
     RTChecks.Bounds.try_emplace(Base, SE->getAddExpr(BaseSC, Off.first),
                                 SE->getAddExpr(BaseSC, Off.second));
   }
-  // Every involved base must contribute at least one bounded access, and its
-  // bounds must be expandable at the guard (so the checks only reference values
-  // available in the header).
-  SCEVExpander Exp(*SE, "slp.rtcheck");
-  // The guard is emitted after the header PHIs, so validate expandability at
-  // the first non-PHI: only values available in the header (PHIs, arguments,
-  // values defined before the block) dominate that point.
-  Instruction *GuardPt = &*BB->getFirstNonPHIIt();
-  if (any_of(Bases, [&](const Value *Base) {
-        auto *It = RTChecks.Bounds.find(Base);
-        return It == RTChecks.Bounds.end() ||
-               !Exp.isSafeToExpandAt(It->second.first, GuardPt) ||
-               !Exp.isSafeToExpandAt(It->second.second, GuardPt);
-      }))
-    return false;
 
-  // No SSA value defined in the body may escape the versioned region: that
-  // would require a merge PHI in the continuation block, which is not yet
-  // supported. A use by the terminator counts as an escape because the
-  // terminator is moved into the continuation block.
+  // SSA values defined in the body and used outside of it are merged from both
+  // paths in the continuation block, so they must stay scalar. Uses by the
+  // terminator (moved into the continuation block) and by the block's PHIs
+  // (via the loop latch) count as outside.
   for (Instruction &I : *BB) {
-    if (isa<PHINode>(&I) || I.isTerminator())
+    if (isa<PHINode>(&I) || I.isTerminator() || !isVectorized(&I))
       continue;
     if (any_of(I.users(), [&](User *U) {
           auto *UI = dyn_cast<Instruction>(U);
-          return !UI || UI->getParent() != BB || UI->isTerminator();
+          return !UI || UI->getParent() != BB || isa<PHINode>(UI) ||
+                 UI->isTerminator();
         }))
       return false;
   }
@@ -25682,6 +25671,21 @@ void BoUpSLP::versionBlocksForRuntimeChecks() {
   for (BasicBlock *Succ : successors(Term))
     Succ->replacePhiUsesWith(BB, Tail);
 
+  // Merge the body values used outside of the versioned region (by the
+  // terminator, across the loop latch or in other blocks) from both paths. The
+  // SCEVs cached for the outside users refer to the body values, which no
+  // longer dominate them.
+  for (Instruction *I : RTOrigBodyOrder) {
+    if (!I->isUsedOutsideOfBlock(VecBB))
+      continue;
+    SE->forgetValue(I);
+    PHINode *Merge = PHINode::Create(I->getType(), 2, I->getName() + ".rtmerge",
+                                     Tail->getFirstNonPHIIt());
+    I->replaceUsesOutsideBlock(Merge, VecBB);
+    Merge->addIncoming(I, VecBB);
+    Merge->addIncoming(VMap.lookup(I), ScalarBB);
+  }
+
   // Keep the dominator tree valid for the remainder of the run.
   if (DT) {
     DomTreeUpdater DTU(*DT, DomTreeUpdater::UpdateStrategy::Eager);
@@ -25704,6 +25708,11 @@ void BoUpSLP::versionBlocksForRuntimeChecks() {
     // instructions across blocks for nodes emitted after versioning.
     DT->updateDFSNumbers();
   }
+  // Keep the loop info valid for the remainder of the run: the new blocks
+  // belong to the loop of the versioned block.
+  if (Loop *L = LI->getLoopFor(BB))
+    for (BasicBlock *NewBB : {VecBB, ScalarBB, Tail})
+      L->addBasicBlockToLoop(NewBB, *LI);
 
   // Record the checked base pairs for the fast-path block so that subsequent
   // vectorization there can reuse this guard instead of versioning again.
@@ -29160,7 +29169,7 @@ SLPVectorizerPass::vectorizeStoreChain(ArrayRef<Value *> Chain, BoUpSLP &R,
   if (R.runtimeChecksFailedForBlock(BB))
     return Res;
   // The retry rebuilds the whole tree; also skip it when the chain's block can
-  // never be versioned (e.g. it is inside a loop).
+  // never be versioned (e.g. the function is optimized for size).
   if (!R.canVersionBlockForRuntimeChecks(BB)) {
     R.markRuntimeChecksFailedForBlock(BB);
     return Res;
diff --git a/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll b/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll
index 92db6896d8f42..9af4179a24940 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll
@@ -1630,9 +1630,33 @@ define void @versioned_loop_block(ptr %dst, i64 %stride, ptr %src, i64 %n) {
 ; CHECK-NEXT:  [[LOOP_RTVEC:.*]]:
 ; CHECK-NEXT:    br label %[[LOOP_RTCONT:.*]]
 ; CHECK:       [[LOOP_RTCONT]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[LOOP_RTVEC]] ], [ [[IV_NEXT_SCALAR:%.*]], %[[LOOP_RTCONT]] ]
-; CHECK-NEXT:    [[D:%.*]] = phi ptr [ [[DST]], %[[LOOP_RTVEC]] ], [ [[D_NEXT_SCALAR:%.*]], %[[LOOP_RTCONT]] ]
-; CHECK-NEXT:    [[S:%.*]] = phi ptr [ [[SRC]], %[[LOOP_RTVEC]] ], [ [[S_NEXT_SCALAR:%.*]], %[[LOOP_RTCONT]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[LOOP_RTVEC]] ], [ [[IV_NEXT_RTMERGE:%.*]], %[[EXIT:.*]] ]
+; CHECK-NEXT:    [[D:%.*]] = phi ptr [ [[DST]], %[[LOOP_RTVEC]] ], [ [[D_NEXT_RTMERGE:%.*]], %[[EXIT]] ]
+; CHECK-NEXT:    [[S:%.*]] = phi ptr [ [[SRC]], %[[LOOP_RTVEC]] ], [ [[S_NEXT_RTMERGE:%.*]], %[[EXIT]] ]
+; CHECK-NEXT:    [[S8:%.*]] = ptrtoaddr ptr [[S]] to i64
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[S8]], 16
+; CHECK-NEXT:    [[D9:%.*]] = ptrtoaddr ptr [[D]] to i64
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[D9]], 8
+; CHECK-NEXT:    [[RT_BOUND0:%.*]] = icmp ult i64 [[D9]], [[TMP0]]
+; CHECK-NEXT:    [[RT_BOUND1:%.*]] = icmp ult i64 [[S8]], [[TMP1]]
+; CHECK-NEXT:    [[RT_CONFLICT:%.*]] = and i1 [[RT_BOUND0]], [[RT_BOUND1]]
+; CHECK-NEXT:    [[RT_GUARD:%.*]] = freeze i1 [[RT_CONFLICT]]
+; CHECK-NEXT:    br i1 [[RT_GUARD]], label %[[LOOP_RTSCALAR:.*]], label %[[LOOP_RTVEC1:.*]], !prof [[PROF0]]
+; CHECK:       [[EXIT1:.*]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[LOOP_RTVEC1]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = load <8 x i16>, ptr [[S]], align 2
+; CHECK-NEXT:    [[TMP3:%.*]] = ashr <8 x i16> [[TMP2]], splat (i16 8)
+; CHECK-NEXT:    [[TMP4:%.*]] = sub <8 x i16> [[TMP2]], [[TMP3]]
+; CHECK-NEXT:    [[TMP5:%.*]] = lshr <8 x i16> [[TMP4]], splat (i16 6)
+; CHECK-NEXT:    [[TMP6:%.*]] = trunc <8 x i16> [[TMP5]] to <8 x i8>
+; CHECK-NEXT:    store <8 x i8> [[TMP6]], ptr [[D]], align 1
+; CHECK-NEXT:    [[D_NEXT:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[STRIDE]]
+; CHECK-NEXT:    [[S_NEXT:%.*]] = getelementptr inbounds nuw i8, ptr [[S]], i64 16
+; CHECK-NEXT:    [[IV_NEXT:%.*]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT:    [[COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[LOOP_RTSCALAR]]:
 ; CHECK-NEXT:    [[L0_SCALAR:%.*]] = load i16, ptr [[S]], align 2
 ; CHECK-NEXT:    [[A0_SCALAR:%.*]] = ashr i16 [[L0_SCALAR]], 8
 ; CHECK-NEXT:    [[B0_SCALAR:%.*]] = sub i16 [[L0_SCALAR]], [[A0_SCALAR]]
@@ -1695,13 +1719,17 @@ define void @versioned_loop_block(ptr %dst, i64 %stride, ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[C7_SCALAR:%.*]] = lshr i16 [[B7_SCALAR]], 6
 ; CHECK-NEXT:    [[T7_SCALAR:%.*]] = trunc i16 [[C7_SCALAR]] to i8
 ; CHECK-NEXT:    store i8 [[T7_SCALAR]], ptr [[D7_SCALAR]], align 1
-; CHECK-NEXT:    [[D_NEXT_SCALAR]] = getelementptr inbounds i8, ptr [[D]], i64 [[STRIDE]]
-; CHECK-NEXT:    [[S_NEXT_SCALAR]] = getelementptr inbounds nuw i8, ptr [[S]], i64 16
-; CHECK-NEXT:    [[IV_NEXT_SCALAR]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT:    [[D_NEXT_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[STRIDE]]
+; CHECK-NEXT:    [[S_NEXT_SCALAR:%.*]] = getelementptr inbounds nuw i8, ptr [[S]], i64 16
+; CHECK-NEXT:    [[IV_NEXT_SCALAR:%.*]] = add nuw i64 [[IV]], 1
 ; CHECK-NEXT:    [[COND_SCALAR:%.*]] = icmp eq i64 [[IV_NEXT_SCALAR]], [[N]]
-; CHECK-NEXT:    br i1 [[COND_SCALAR]], label %[[EXIT:.*]], label %[[LOOP_RTCONT]]
+; CHECK-NEXT:    br label %[[EXIT]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    ret void
+; CHECK-NEXT:    [[D_NEXT_RTMERGE]] = phi ptr [ [[D_NEXT]], %[[LOOP_RTVEC1]] ], [ [[D_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT:    [[S_NEXT_RTMERGE]] = phi ptr [ [[S_NEXT]], %[[LOOP_RTVEC1]] ], [ [[S_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT:    [[IV_NEXT_RTMERGE]] = phi i64 [ [[IV_NEXT]], %[[LOOP_RTVEC1]] ], [ [[IV_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT:    [[COND_RTMERGE:%.*]] = phi i1 [ [[COND]], %[[LOOP_RTVEC1]] ], [ [[COND_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT:    br i1 [[COND_RTMERGE]], label %[[EXIT1]], label %[[LOOP_RTCONT]]
 ;
 ; NOCHK-LABEL: define void @versioned_loop_block(
 ; NOCHK-SAME: ptr [[DST:%.*]], i64 [[STRIDE:%.*]], ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR1]] {



More information about the llvm-commits mailing list