[llvm] [LAA] Fix off-by-EltSize in negative-step deref bounds check (PR #211964)

Aleksandr Popov via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 7 13:46:17 PDT 2026


https://github.com/aleks-tmb updated https://github.com/llvm/llvm-project/pull/211964

>From 07667fca8e458ff4681b97b6b3355314307f3d35 Mon Sep 17 00:00:00 2001
From: Aleksandr Popov <apopov at azul.com>
Date: Mon, 10 Aug 2026 21:19:15 +0000
Subject: [PATCH 1/2] Precommit test

---
 .../reverse-loop-dynamic-length.ll            | 55 +++++++++++++++++++
 1 file changed, 55 insertions(+)
 create mode 100644 llvm/test/Transforms/LoopVectorize/reverse-loop-dynamic-length.ll

diff --git a/llvm/test/Transforms/LoopVectorize/reverse-loop-dynamic-length.ll b/llvm/test/Transforms/LoopVectorize/reverse-loop-dynamic-length.ll
new file mode 100644
index 0000000000000..c898b478faa3c
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/reverse-loop-dynamic-length.ll
@@ -0,0 +1,55 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "scalar.ph:" --version 6
+; RUN: opt -p loop-vectorize -force-vector-width=4 -S %s | FileCheck %s
+
+; TODO: LAA should recognise that the AR fits within the deref region 
+; and produce tight bounds, allowing the early-exit loop to be vectorized
+; without runtime memory checks.
+define ptr @reverse_reaches_base(i64 %length, ptr %ptr) {
+; CHECK-LABEL: define ptr @reverse_reaches_base(
+; CHECK-SAME: i64 [[LENGTH:%.*]], ptr [[PTR:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[NULL_CHECK:%.*]] = icmp eq i64 [[LENGTH]], 0
+; CHECK-NEXT:    br i1 [[NULL_CHECK]], label %[[EXIT:.*]], label %[[PREHEADER:.*]]
+; CHECK:       [[PREHEADER]]:
+; CHECK-NEXT:    call void @llvm.assume(i1 true) [ "dereferenceable"(ptr [[PTR]], i64 [[LENGTH]]) ]
+; CHECK-NEXT:    [[START:%.*]] = sub i64 [[LENGTH]], 1
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = phi i64 [ [[IV_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ], [ [[START]], %[[PREHEADER]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr i8, ptr [[PTR]], i64 [[TMP1]]
+; CHECK-NEXT:    [[ELEMENT:%.*]] = load i8, ptr [[TMP2]], align 1
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i8 [[ELEMENT]], 0
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
+; CHECK:       [[VECTOR_BODY_INTERIM]]:
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[TMP1]], -1
+; CHECK-NEXT:    [[RANGE_CHECK:%.*]] = icmp ne i64 [[TMP1]], 0
+; CHECK-NEXT:    br i1 [[RANGE_CHECK]], label %[[VECTOR_BODY]], label %[[VECTOR_EARLY_EXIT]]
+; CHECK:       [[VECTOR_EARLY_EXIT]]:
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret ptr null
+;
+entry:
+  %null_check = icmp eq i64 %length, 0
+  br i1 %null_check, label %exit, label %preheader
+
+preheader:
+  call void @llvm.assume(i1 true) [ "dereferenceable"(ptr %ptr, i64 %length) ]
+  %start = sub i64 %length, 1
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %iv.next, %latch ], [ %start, %preheader ]
+  %element_gep = getelementptr i8, ptr %ptr, i64 %iv
+  %element = load i8, ptr %element_gep, align 1
+  %found_check = icmp eq i8 %element, 0
+  br i1 %found_check, label %exit, label %latch
+
+latch:
+  %iv.next = add i64 %iv, -1
+  %range_check = icmp ne i64 %iv, 0
+  br i1 %range_check, label %loop, label %exit
+
+exit:
+  ret ptr null
+}

>From 4c03b3c6b320be57576caecd4938a23856d73d81 Mon Sep 17 00:00:00 2001
From: Aleksandr Popov <apopov at azul.com>
Date: Mon, 27 Jul 2026 00:08:52 +0000
Subject: [PATCH 2/2] [LAA] Fix off-by-EltSize bounds in reverse-loop deref
 no-wrap check

evaluatePtrAddRecAtMaxBTCWillNotWrap used AR->getStart() as the
lowest accessed address; for a negative step it is the *highest*.
Both safety checks on the reverse-loop branch inherited this and
were off by EltSize:

  * No-underflow: added an extra EltSize of slack, rejecting
    reverse loops whose last iteration lands on the base pointer.
  * Deref-end: dropped the size of the top access, accepting
    loops whose top read spills past DerefBytes.

Pick LowestAddr per step direction (AR->getStart() vs.
AR->evaluateAtIteration(MaxBTC, SE)). The range algebra downstream
becomes direction-agnostic and the per-direction MaxOffset branch
collapses into a single expression.
---
 llvm/lib/Analysis/LoopAccessAnalysis.cpp      | 53 +++++++++----------
 .../negative-step-deref-off-by-eltsize.ll     | 12 ++---
 .../reverse-loop-dynamic-length.ll            | 35 ++++++++----
 3 files changed, 54 insertions(+), 46 deletions(-)

diff --git a/llvm/lib/Analysis/LoopAccessAnalysis.cpp b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
index 2d81fb270df63..86e742061e992 100644
--- a/llvm/lib/Analysis/LoopAccessAnalysis.cpp
+++ b/llvm/lib/Analysis/LoopAccessAnalysis.cpp
@@ -206,6 +206,16 @@ static const SCEV *mulSCEVNoOverflow(const SCEV *A, const SCEV *B,
 
 /// Return true, if evaluating \p AR at \p MaxBTC cannot wrap, because \p AR at
 /// \p MaxBTC is guaranteed inbounds of the accessed object.
+///
+/// The accessed byte range is [LowestOffset, LowestOffset + AccessedBytes),
+/// where
+///   AccessedBytes = MaxBTC * |Step| + EltSize,
+///   LowestOffset  = smallest byte offset from StartPtr any iteration reaches.
+///
+/// The function returns true only when both safety invariants hold, regardless
+/// of step direction:
+///   1. LowestOffset >= 0                          (no access below StartPtr)
+///   2. LowestOffset + AccessedBytes <= DerefBytes (no access past the region)
 static bool evaluatePtrAddRecAtMaxBTCWillNotWrap(
     const SCEVAddRecExpr *AR, const SCEV *MaxBTC, const SCEV *EltSize,
     ScalarEvolution &SE, const DataLayout &DL, DominatorTree *DT,
@@ -264,15 +274,15 @@ static bool evaluatePtrAddRecAtMaxBTCWillNotWrap(
   Step = SE.getNoopOrSignExtend(Step, WiderTy);
   MaxBTC = SE.getNoopOrZeroExtend(MaxBTC, WiderTy);
 
-  // For the computations below, make sure they don't unsigned wrap.
-  // FIXME: for a negative step the lowest accessed address is not
-  // AR->getStart() but AR->evaluateAtIteration(MaxBTC, SE); the check below
-  // therefore compares StartPtr against the highest accessed address instead
-  // of the lowest.
-  if (!SE.isKnownPredicate(CmpInst::ICMP_UGE, AR->getStart(), StartPtr))
+  const SCEV *LowestAddr = IsKnownNonNegative
+                               ? static_cast<const SCEV *>(AR->getStart())
+                               : AR->evaluateAtIteration(MaxBTC, SE);
+  // Lower-bound safety check: the lowest accessed address must not fall below
+  // StartPtr.
+  if (!SE.isKnownPredicate(CmpInst::ICMP_UGE, LowestAddr, StartPtr))
     return false;
-  const SCEV *StartOffset = SE.getNoopOrZeroExtend(
-      SE.getMinusSCEV(AR->getStart(), StartPtr), WiderTy);
+  const SCEV *LowestOffset =
+      SE.getNoopOrZeroExtend(SE.getMinusSCEV(LowestAddr, StartPtr), WiderTy);
 
   if (!LoopGuards)
     LoopGuards.emplace(ScalarEvolution::LoopGuards::collect(AR->getLoop(), SE));
@@ -301,27 +311,12 @@ static bool evaluatePtrAddRecAtMaxBTCWillNotWrap(
   if (!AccessedBytes)
     return false;
 
-  // Compute MaxOffset per direction: exclusive upper offset of the
-  // accessed range.
-  const SCEV *MaxOffset;
-  if (IsKnownNonNegative) {
-    MaxOffset = addSCEVNoOverflow(StartOffset, AccessedBytes, SE);
-    if (!MaxOffset)
-      return false;
-    DerefBytesSCEV = SE.applyLoopGuards(DerefBytesSCEV, *LoopGuards);
-  } else {
-    // FIXME: two independent off-by-EltSize bugs on this branch:
-    //  1. StartOffset here is actually the HIGHEST offset, because it is
-    //     computed from AR->getStart() rather than
-    //     AR->evaluateAtIteration(MaxBTC, SE) (see FIXME above).
-    //  2. The lower check is over-strict by EltSize and the upper is
-    //     under-counted by EltSize.
-    assert(SE.isKnownNegative(Step) && "must be known negative");
-    if (!SE.isKnownPredicate(CmpInst::ICMP_SGE, StartOffset, AccessedBytes))
-      return false;
-    MaxOffset = StartOffset;
-  }
-  // MaxOffset must not exceed the deref-region end.
+  // Exclusive upper offset of the accessed range.
+  const SCEV *MaxOffset = addSCEVNoOverflow(LowestOffset, AccessedBytes, SE);
+  if (!MaxOffset)
+    return false;
+  DerefBytesSCEV = SE.applyLoopGuards(DerefBytesSCEV, *LoopGuards);
+  // Upper-bound safety check: MaxOffset must not exceed the deref-region end.
   return SE.isKnownPredicate(CmpInst::ICMP_ULE, MaxOffset, DerefBytesSCEV);
 }
 
diff --git a/llvm/test/Analysis/LoopAccessAnalysis/negative-step-deref-off-by-eltsize.ll b/llvm/test/Analysis/LoopAccessAnalysis/negative-step-deref-off-by-eltsize.ll
index b9f1857e40137..ec1d790c6a664 100644
--- a/llvm/test/Analysis/LoopAccessAnalysis/negative-step-deref-off-by-eltsize.ll
+++ b/llvm/test/Analysis/LoopAccessAnalysis/negative-step-deref-off-by-eltsize.ll
@@ -4,7 +4,7 @@
 ; Reverse loop loading 4 i32 elements whose access range exactly fills the
 ; dereferenceable region (deref(16), reads bytes [0, 16)).
 ;
-; TODO: LAA should recognise that this AR fits within the deref
+; LAA should recognise that this AR fits within the deref
 ; region and produce tight bounds (Low: %A, High: %A + 16).
 ;
 ; Pseudocode:
@@ -28,10 +28,10 @@ define void @reverse_reaches_base(ptr dereferenceable(16) %A, ptr dereferenceabl
 ; CHECK-NEXT:          %gep.A = getelementptr inbounds i32, ptr %A, i64 %iv
 ; CHECK-NEXT:      Grouped accesses:
 ; CHECK-NEXT:        Group GRP0:
-; CHECK-NEXT:          (Low: null High: (16 + %B)<nuw>)
+; CHECK-NEXT:          (Low: %B High: (16 + %B)<nuw>)
 ; CHECK-NEXT:            Member: {(12 + %B)<nuw>,+,-4}<nw><%loop>
 ; CHECK-NEXT:        Group GRP1:
-; CHECK-NEXT:          (Low: null High: (16 + %A)<nuw>)
+; CHECK-NEXT:          (Low: %A High: (16 + %A)<nuw>)
 ; CHECK-NEXT:            Member: {(12 + %A)<nuw>,+,-4}<nw><%loop>
 ; CHECK-EMPTY:
 ; CHECK-NEXT:      Non vectorizable stores to invariant address were not found in loop.
@@ -67,7 +67,7 @@ exit.done:
 ; The top i32 read at byte 13 covers [13, 17), but deref(16) only
 ; guarantees [0, 16) — bytes at/after 16 may or may not be dereferenceable.
 ;
-; TODO: LAA must not assume the AR fits in the deref region and should
+; LAA must not assume the AR fits in the deref region and should
 ; fall back to the wide low bound.
 ;
 ; Pseudocode:
@@ -90,10 +90,10 @@ define void @reverse_top_spills(ptr dereferenceable(16) %A, ptr dereferenceable(
 ; CHECK-NEXT:          %gep.A = getelementptr inbounds i8, ptr %A, i64 %iv
 ; CHECK-NEXT:      Grouped accesses:
 ; CHECK-NEXT:        Group GRP0:
-; CHECK-NEXT:          (Low: (5 + %B)<nuw> High: (17 + %B))
+; CHECK-NEXT:          (Low: null High: (17 + %B))
 ; CHECK-NEXT:            Member: {(13 + %B)<nuw>,+,-4}<nw><%loop>
 ; CHECK-NEXT:        Group GRP1:
-; CHECK-NEXT:          (Low: (5 + %A)<nuw> High: (17 + %A))
+; CHECK-NEXT:          (Low: null High: (17 + %A))
 ; CHECK-NEXT:            Member: {(13 + %A)<nuw>,+,-4}<nw><%loop>
 ; CHECK-EMPTY:
 ; CHECK-NEXT:      Non vectorizable stores to invariant address were not found in loop.
diff --git a/llvm/test/Transforms/LoopVectorize/reverse-loop-dynamic-length.ll b/llvm/test/Transforms/LoopVectorize/reverse-loop-dynamic-length.ll
index c898b478faa3c..91d68a0ad297c 100644
--- a/llvm/test/Transforms/LoopVectorize/reverse-loop-dynamic-length.ll
+++ b/llvm/test/Transforms/LoopVectorize/reverse-loop-dynamic-length.ll
@@ -1,7 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --filter-out-after "scalar.ph:" --version 6
 ; RUN: opt -p loop-vectorize -force-vector-width=4 -S %s | FileCheck %s
 
-; TODO: LAA should recognise that the AR fits within the deref region 
+; LAA should recognise that the AR fits within the deref region
 ; and produce tight bounds, allowing the early-exit loop to be vectorized
 ; without runtime memory checks.
 define ptr @reverse_reaches_base(i64 %length, ptr %ptr) {
@@ -9,25 +9,38 @@ define ptr @reverse_reaches_base(i64 %length, ptr %ptr) {
 ; CHECK-SAME: i64 [[LENGTH:%.*]], ptr [[PTR:%.*]]) {
 ; CHECK-NEXT:  [[ENTRY:.*:]]
 ; CHECK-NEXT:    [[NULL_CHECK:%.*]] = icmp eq i64 [[LENGTH]], 0
-; CHECK-NEXT:    br i1 [[NULL_CHECK]], label %[[EXIT:.*]], label %[[PREHEADER:.*]]
+; CHECK-NEXT:    br i1 [[NULL_CHECK]], [[EXIT:label %.*]], label %[[PREHEADER:.*]]
 ; CHECK:       [[PREHEADER]]:
 ; CHECK-NEXT:    call void @llvm.assume(i1 true) [ "dereferenceable"(ptr [[PTR]], i64 [[LENGTH]]) ]
 ; CHECK-NEXT:    [[START:%.*]] = sub i64 [[LENGTH]], 1
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[LENGTH]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i64 [[LENGTH]], 4
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i64 [[LENGTH]], [[N_MOD_VF]]
+; CHECK-NEXT:    [[TMP0:%.*]] = sub i64 [[START]], [[N_VEC]]
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = phi i64 [ [[IV_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ], [ [[START]], %[[PREHEADER]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY_INTERIM:.*]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = sub i64 [[START]], [[INDEX]]
 ; CHECK-NEXT:    [[TMP2:%.*]] = getelementptr i8, ptr [[PTR]], i64 [[TMP1]]
-; CHECK-NEXT:    [[ELEMENT:%.*]] = load i8, ptr [[TMP2]], align 1
-; CHECK-NEXT:    [[TMP6:%.*]] = icmp eq i8 [[ELEMENT]], 0
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr i8, ptr [[TMP2]], i64 -3
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
+; CHECK-NEXT:    [[REVERSE:%.*]] = shufflevector <4 x i8> [[WIDE_LOAD]], <4 x i8> poison, <4 x i32> <i32 3, i32 2, i32 1, i32 0>
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq <4 x i8> [[REVERSE]], zeroinitializer
+; CHECK-NEXT:    [[TMP5:%.*]] = freeze <4 x i1> [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP6]], label %[[VECTOR_EARLY_EXIT:.*]], label %[[VECTOR_BODY_INTERIM]]
 ; CHECK:       [[VECTOR_BODY_INTERIM]]:
-; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[TMP1]], -1
-; CHECK-NEXT:    [[RANGE_CHECK:%.*]] = icmp ne i64 [[TMP1]], 0
-; CHECK-NEXT:    br i1 [[RANGE_CHECK]], label %[[VECTOR_BODY]], label %[[VECTOR_EARLY_EXIT]]
+; CHECK-NEXT:    br i1 [[TMP7]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i64 [[LENGTH]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], [[EXIT_LOOPEXIT:label %.*]], label %[[SCALAR_PH]]
 ; CHECK:       [[VECTOR_EARLY_EXIT]]:
-; CHECK-NEXT:    br label %[[EXIT]]
-; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    ret ptr null
+; CHECK-NEXT:    br [[EXIT_LOOPEXIT]]
+; CHECK:       [[SCALAR_PH]]:
 ;
 entry:
   %null_check = icmp eq i64 %length, 0



More information about the llvm-commits mailing list