[llvm] fa10e63 - [SLP]Support runtime alias checks versioning for blocks in loops

via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 29 04:05:48 PDT 2026


Author: Alexey Bataev
Date: 2026-09-29T07:05:42-04:00
New Revision: fa10e634e2be9c16a7fdf01aefe0ed45ab6d1b47

URL: https://github.com/llvm/llvm-project/commit/fa10e634e2be9c16a7fdf01aefe0ed45ab6d1b47
DIFF: https://github.com/llvm/llvm-project/commit/fa10e634e2be9c16a7fdf01aefe0ed45ab6d1b47.diff

LOG: [SLP]Support runtime alias checks versioning for blocks in loops

Allow versioning of blocks inside loops, e.g. fully unrolled inner loop,
whose pointers are advanced by the loop header PHIs. Accept header PHIs
as base objects, use the base values directly in the bounds instead of
recomputing loop-variant bases from their recurrences.

Fixes #49896

Assisted-by: Cursor

Reviewers: RKSimon, bababuck

Pull Request: https://github.com/llvm/llvm-project/pull/226955

Added: 
    

Modified: 
    llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
    llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll

Removed: 
    


################################################################################
diff  --git a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
index 7e23795f398a7..e0dd13ce5c2a8 100644
--- a/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
+++ b/llvm/lib/Transforms/Vectorize/SLPVectorizer.cpp
@@ -541,10 +541,10 @@ class slpvectorizer::BoUpSLP {
   void captureRuntimeCheckBodySnapshot();
 
   /// Returns true if \p BB satisfies the block-level preconditions for runtime
-  /// alias check versioning (straight-line, outside any loop, duplicable, not a
-  /// scalar fallback, function not optimized for size). These checks do not
-  /// depend on the collected checks, so they can gate the (expensive)
-  /// optimistic retry before any tree is rebuilt.
+  /// alias check versioning (straight-line, duplicable, not a scalar fallback,
+  /// function not optimized for size). These checks do not depend on the
+  /// collected checks, so they can gate the (expensive) optimistic retry before
+  /// any tree is rebuilt.
   bool canVersionBlockForRuntimeChecks(BasicBlock *BB) const;
 
   /// Returns true if the runtime alias checks can be safely emitted to guard
@@ -25484,10 +25484,6 @@ bool BoUpSLP::canVersionBlockForRuntimeChecks(BasicBlock *BB) const {
     return false;
   if (!DT->isReachableFromEntry(BB))
     return false;
-  // Versioning duplicates the block body; only straight-line code outside any
-  // loop is handled for now, to avoid LoopInfo and region updates.
-  if (LI->getLoopFor(BB))
-    return false;
   // Never version a scalar fallback block: it is the safe, original-order copy
   // taken when aliasing is detected and must stay scalar.
   if (ScalarFallbackBlocks.contains(BB))
@@ -25543,9 +25539,9 @@ bool BoUpSLP::canVersionForRuntimeChecks() {
     Bases.insert(P.second);
   }
   // Every base object must be available in the (PHI-only) header where the
-  // guard branch is emitted.
+  // guard branch is emitted, i.e. be one of its PHIs or dominate it.
   if (any_of(make_isa_range<Instruction>(Bases), [&](const Instruction *I) {
-        return I->getParent() == BB || !DT->dominates(I, BB);
+        return I->getParent() == BB ? !isa<PHINode>(I) : !DT->dominates(I, BB);
       }))
     return false;
 
@@ -25602,40 +25598,32 @@ bool BoUpSLP::canVersionForRuntimeChecks() {
       It->second.second = SE->getUMaxExpr(It->second.second, EndOff);
     }
   }
+  // Every involved base must contribute at least one bounded access.
+  if (OffBounds.size() != Bases.size())
+    return false;
   // Materialize the absolute [Low, High) = base + [minOff, maxEndOff). The
   // offsets are constants, so each bound is a base-plus-constant expression
   // that expands to a single add off the base address (no runtime umin/umax).
+  // The base is used as an opaque value, so the bounds only reference the base
+  // itself, available at the guard, and a loop-variant base (e.g. a header PHI)
+  // is reused rather than recomputed from its recurrence.
   for (const auto &[Base, Off] : OffBounds) {
-    const SCEV *BaseSC = SE->getPtrToAddrExpr(SE->getSCEV(Base));
+    const SCEV *BaseSC = SE->getPtrToAddrExpr(SE->getUnknown(Base));
     RTChecks.Bounds.try_emplace(Base, SE->getAddExpr(BaseSC, Off.first),
                                 SE->getAddExpr(BaseSC, Off.second));
   }
-  // Every involved base must contribute at least one bounded access, and its
-  // bounds must be expandable at the guard (so the checks only reference values
-  // available in the header).
-  SCEVExpander Exp(*SE, "slp.rtcheck");
-  // The guard is emitted after the header PHIs, so validate expandability at
-  // the first non-PHI: only values available in the header (PHIs, arguments,
-  // values defined before the block) dominate that point.
-  Instruction *GuardPt = &*BB->getFirstNonPHIIt();
-  if (any_of(Bases, [&](const Value *Base) {
-        auto *It = RTChecks.Bounds.find(Base);
-        return It == RTChecks.Bounds.end() ||
-               !Exp.isSafeToExpandAt(It->second.first, GuardPt) ||
-               !Exp.isSafeToExpandAt(It->second.second, GuardPt);
-      }))
-    return false;
 
-  // No SSA value defined in the body may escape the versioned region: that
-  // would require a merge PHI in the continuation block, which is not yet
-  // supported. A use by the terminator counts as an escape because the
-  // terminator is moved into the continuation block.
-  for (Instruction &I : make_filter_range(*BB, [](Instruction &I) {
-         return !isa<PHINode>(&I) && !I.isTerminator();
+  // SSA values defined in the body and used outside of it are merged from both
+  // paths in the continuation block, so they must stay scalar. Uses by the
+  // terminator (moved into the continuation block) and by the block's PHIs
+  // (via the loop latch) count as outside.
+  for (Instruction &I : make_filter_range(*BB, [&](Instruction &I) {
+         return !isa<PHINode>(&I) && !I.isTerminator() && isVectorized(&I);
        })) {
     if (any_of(I.users(), [&](User *U) {
           auto *UI = dyn_cast<Instruction>(U);
-          return !UI || UI->getParent() != BB || UI->isTerminator();
+          return !UI || UI->getParent() != BB || isa<PHINode>(UI) ||
+                 UI->isTerminator();
         }))
       return false;
   }
@@ -25797,6 +25785,21 @@ void BoUpSLP::versionBlocksForRuntimeChecks() {
   for (BasicBlock *Succ : successors(Term))
     Succ->replacePhiUsesWith(BB, Tail);
 
+  // Merge the body values used outside of the versioned region (by the
+  // terminator, across the loop latch or in other blocks) from both paths. The
+  // SCEVs cached for the outside users refer to the body values, which no
+  // longer dominate them.
+  for (Instruction *I : RTOrigBodyOrder) {
+    if (!I->isUsedOutsideOfBlock(VecBB))
+      continue;
+    SE->forgetValue(I);
+    PHINode *Merge = PHINode::Create(I->getType(), 2, I->getName() + ".rtmerge",
+                                     Tail->getFirstNonPHIIt());
+    I->replaceUsesOutsideBlock(Merge, VecBB);
+    Merge->addIncoming(I, VecBB);
+    Merge->addIncoming(VMap.lookup(I), ScalarBB);
+  }
+
   // Keep the dominator tree valid for the remainder of the run.
   if (DT) {
     DomTreeUpdater DTU(*DT, DomTreeUpdater::UpdateStrategy::Eager);
@@ -25819,6 +25822,11 @@ void BoUpSLP::versionBlocksForRuntimeChecks() {
     // instructions across blocks for nodes emitted after versioning.
     DT->updateDFSNumbers();
   }
+  // Keep the loop info valid for the remainder of the run: the new blocks
+  // belong to the loop of the versioned block.
+  if (Loop *L = LI->getLoopFor(BB))
+    for (BasicBlock *NewBB : {VecBB, ScalarBB, Tail})
+      L->addBasicBlockToLoop(NewBB, *LI);
 
   // Record the checked base pairs for the fast-path block so that subsequent
   // vectorization there can reuse this guard instead of versioning again.
@@ -29282,7 +29290,7 @@ SLPVectorizerPass::vectorizeStoreChain(ArrayRef<Value *> Chain, BoUpSLP &R,
   if (R.runtimeChecksFailedForBlock(BB))
     return Res;
   // The retry rebuilds the whole tree; also skip it when the chain's block can
-  // never be versioned (e.g. it is inside a loop).
+  // never be versioned (e.g. the function is optimized for size).
   if (!R.canVersionBlockForRuntimeChecks(BB)) {
     R.markRuntimeChecksFailedForBlock(BB);
     return Res;

diff  --git a/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll b/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll
index 92db6896d8f42..9af4179a24940 100644
--- a/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll
+++ b/llvm/test/Transforms/SLPVectorizer/X86/runtime-alias-checks.ll
@@ -1630,9 +1630,33 @@ define void @versioned_loop_block(ptr %dst, i64 %stride, ptr %src, i64 %n) {
 ; CHECK-NEXT:  [[LOOP_RTVEC:.*]]:
 ; CHECK-NEXT:    br label %[[LOOP_RTCONT:.*]]
 ; CHECK:       [[LOOP_RTCONT]]:
-; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[LOOP_RTVEC]] ], [ [[IV_NEXT_SCALAR:%.*]], %[[LOOP_RTCONT]] ]
-; CHECK-NEXT:    [[D:%.*]] = phi ptr [ [[DST]], %[[LOOP_RTVEC]] ], [ [[D_NEXT_SCALAR:%.*]], %[[LOOP_RTCONT]] ]
-; CHECK-NEXT:    [[S:%.*]] = phi ptr [ [[SRC]], %[[LOOP_RTVEC]] ], [ [[S_NEXT_SCALAR:%.*]], %[[LOOP_RTCONT]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[LOOP_RTVEC]] ], [ [[IV_NEXT_RTMERGE:%.*]], %[[EXIT:.*]] ]
+; CHECK-NEXT:    [[D:%.*]] = phi ptr [ [[DST]], %[[LOOP_RTVEC]] ], [ [[D_NEXT_RTMERGE:%.*]], %[[EXIT]] ]
+; CHECK-NEXT:    [[S:%.*]] = phi ptr [ [[SRC]], %[[LOOP_RTVEC]] ], [ [[S_NEXT_RTMERGE:%.*]], %[[EXIT]] ]
+; CHECK-NEXT:    [[S8:%.*]] = ptrtoaddr ptr [[S]] to i64
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[S8]], 16
+; CHECK-NEXT:    [[D9:%.*]] = ptrtoaddr ptr [[D]] to i64
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[D9]], 8
+; CHECK-NEXT:    [[RT_BOUND0:%.*]] = icmp ult i64 [[D9]], [[TMP0]]
+; CHECK-NEXT:    [[RT_BOUND1:%.*]] = icmp ult i64 [[S8]], [[TMP1]]
+; CHECK-NEXT:    [[RT_CONFLICT:%.*]] = and i1 [[RT_BOUND0]], [[RT_BOUND1]]
+; CHECK-NEXT:    [[RT_GUARD:%.*]] = freeze i1 [[RT_CONFLICT]]
+; CHECK-NEXT:    br i1 [[RT_GUARD]], label %[[LOOP_RTSCALAR:.*]], label %[[LOOP_RTVEC1:.*]], !prof [[PROF0]]
+; CHECK:       [[EXIT1:.*]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[LOOP_RTVEC1]]:
+; CHECK-NEXT:    [[TMP2:%.*]] = load <8 x i16>, ptr [[S]], align 2
+; CHECK-NEXT:    [[TMP3:%.*]] = ashr <8 x i16> [[TMP2]], splat (i16 8)
+; CHECK-NEXT:    [[TMP4:%.*]] = sub <8 x i16> [[TMP2]], [[TMP3]]
+; CHECK-NEXT:    [[TMP5:%.*]] = lshr <8 x i16> [[TMP4]], splat (i16 6)
+; CHECK-NEXT:    [[TMP6:%.*]] = trunc <8 x i16> [[TMP5]] to <8 x i8>
+; CHECK-NEXT:    store <8 x i8> [[TMP6]], ptr [[D]], align 1
+; CHECK-NEXT:    [[D_NEXT:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[STRIDE]]
+; CHECK-NEXT:    [[S_NEXT:%.*]] = getelementptr inbounds nuw i8, ptr [[S]], i64 16
+; CHECK-NEXT:    [[IV_NEXT:%.*]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT:    [[COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[LOOP_RTSCALAR]]:
 ; CHECK-NEXT:    [[L0_SCALAR:%.*]] = load i16, ptr [[S]], align 2
 ; CHECK-NEXT:    [[A0_SCALAR:%.*]] = ashr i16 [[L0_SCALAR]], 8
 ; CHECK-NEXT:    [[B0_SCALAR:%.*]] = sub i16 [[L0_SCALAR]], [[A0_SCALAR]]
@@ -1695,13 +1719,17 @@ define void @versioned_loop_block(ptr %dst, i64 %stride, ptr %src, i64 %n) {
 ; CHECK-NEXT:    [[C7_SCALAR:%.*]] = lshr i16 [[B7_SCALAR]], 6
 ; CHECK-NEXT:    [[T7_SCALAR:%.*]] = trunc i16 [[C7_SCALAR]] to i8
 ; CHECK-NEXT:    store i8 [[T7_SCALAR]], ptr [[D7_SCALAR]], align 1
-; CHECK-NEXT:    [[D_NEXT_SCALAR]] = getelementptr inbounds i8, ptr [[D]], i64 [[STRIDE]]
-; CHECK-NEXT:    [[S_NEXT_SCALAR]] = getelementptr inbounds nuw i8, ptr [[S]], i64 16
-; CHECK-NEXT:    [[IV_NEXT_SCALAR]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT:    [[D_NEXT_SCALAR:%.*]] = getelementptr inbounds i8, ptr [[D]], i64 [[STRIDE]]
+; CHECK-NEXT:    [[S_NEXT_SCALAR:%.*]] = getelementptr inbounds nuw i8, ptr [[S]], i64 16
+; CHECK-NEXT:    [[IV_NEXT_SCALAR:%.*]] = add nuw i64 [[IV]], 1
 ; CHECK-NEXT:    [[COND_SCALAR:%.*]] = icmp eq i64 [[IV_NEXT_SCALAR]], [[N]]
-; CHECK-NEXT:    br i1 [[COND_SCALAR]], label %[[EXIT:.*]], label %[[LOOP_RTCONT]]
+; CHECK-NEXT:    br label %[[EXIT]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    ret void
+; CHECK-NEXT:    [[D_NEXT_RTMERGE]] = phi ptr [ [[D_NEXT]], %[[LOOP_RTVEC1]] ], [ [[D_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT:    [[S_NEXT_RTMERGE]] = phi ptr [ [[S_NEXT]], %[[LOOP_RTVEC1]] ], [ [[S_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT:    [[IV_NEXT_RTMERGE]] = phi i64 [ [[IV_NEXT]], %[[LOOP_RTVEC1]] ], [ [[IV_NEXT_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT:    [[COND_RTMERGE:%.*]] = phi i1 [ [[COND]], %[[LOOP_RTVEC1]] ], [ [[COND_SCALAR]], %[[LOOP_RTSCALAR]] ]
+; CHECK-NEXT:    br i1 [[COND_RTMERGE]], label %[[EXIT1]], label %[[LOOP_RTCONT]]
 ;
 ; NOCHK-LABEL: define void @versioned_loop_block(
 ; NOCHK-SAME: ptr [[DST:%.*]], i64 [[STRIDE:%.*]], ptr [[SRC:%.*]], i64 [[N:%.*]]) #[[ATTR1]] {


        


More information about the llvm-commits mailing list