[llvm] [SCEV] Rework wrap-flag-inference for sext-addrec (PR #217362)

Ramkumar Ramachandra via llvm-commits llvm-commits at lists.llvm.org
Thu Aug 27 01:00:15 PDT 2026


https://github.com/artagnon updated https://github.com/llvm/llvm-project/pull/217362

>From 59f735d4c72f72f014e7277534f4da32bebe7e47 Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Wed, 26 Aug 2026 14:44:34 +0100
Subject: [PATCH 1/5] [SCEV] Add test coverage for ext-nusw/nsuw inference

Complete the existing incorrect-nsw test, showing that zext-addrec nusw
is not implied by nw on the pre-inc AR, and that sext-addrec nsuw is not
implied by nw on the pre-inc AR. See also: #217405 and #217362.

Assisted-by: AI
---
 .../ScalarEvolution/ext-addrec-wrap-flags.ll  | 174 ++++++++++++++++++
 .../Analysis/ScalarEvolution/incorrect-nsw.ll |  26 ---
 2 files changed, 174 insertions(+), 26 deletions(-)
 create mode 100644 llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
 delete mode 100644 llvm/test/Analysis/ScalarEvolution/incorrect-nsw.ll

diff --git a/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll b/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
new file mode 100644
index 0000000000000..9e1e1387eead5
--- /dev/null
+++ b/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
@@ -0,0 +1,174 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -disable-output "-passes=print<scalar-evolution>" %s 2>&1 | FileCheck %s
+
+; The sext expression should be {-1,+,-128}<nw>.
+; nw should be inferred correctly.
+define void @sext.nw.pre.inc() {
+; CHECK-LABEL: 'sext.nw.pre.inc'
+; CHECK-NEXT:  Classifying expressions for: @sext.nw.pre.inc
+; CHECK-NEXT:    %i = phi i8 [ -1, %entry ], [ %i.inc, %loop ]
+; CHECK-NEXT:    --> {-1,+,-128}<%loop> U: [-1,-128) S: [-1,-128) Exits: 127 LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %counter = phi i8 [ 0, %entry ], [ %counter.inc, %loop ]
+; CHECK-NEXT:    --> {0,+,1}<nuw><nsw><%loop> U: [0,2) S: [0,2) Exits: 1 LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %i.inc = add i8 %i, -128
+; CHECK-NEXT:    --> {127,+,-128}<%loop> U: [127,0) S: [127,0) Exits: -1 LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %i.sext = sext i8 %i to i16
+; CHECK-NEXT:    --> {-1,+,128}<nw><%loop> U: [-1,128) S: [-1,128) Exits: 127 LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %counter.inc = add i8 %counter, 1
+; CHECK-NEXT:    --> {1,+,1}<nuw><nsw><%loop> U: [1,3) S: [1,3) Exits: 2 LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:  Determining loop execution counts for: @sext.nw.pre.inc
+; CHECK-NEXT:  Loop %loop: backedge-taken count is i8 1
+; CHECK-NEXT:  Loop %loop: constant max backedge-taken count is i8 1
+; CHECK-NEXT:  Loop %loop: symbolic max backedge-taken count is i8 1
+; CHECK-NEXT:  Loop %loop: Trip multiple is 2
+;
+ entry:
+  br label %loop
+
+ loop:
+  %i = phi i8 [ -1, %entry ], [ %i.inc, %loop ]
+  %counter = phi i8 [ 0, %entry ], [ %counter.inc, %loop ]
+  %i.inc = add i8 %i, -128
+  %i.sext = sext i8 %i to i16
+  %counter.inc = add i8 %counter, 1
+  %continue = icmp eq i8 %counter, 1
+  br i1 %continue, label %exit, label %loop
+
+ exit:
+  ret void
+}
+
+
+; The zext expression should be {255,+,-128}<nw>.
+; nw should be inferred correctly.
+define void @zext.nw.pre.inc() {
+; CHECK-LABEL: 'zext.nw.pre.inc'
+; CHECK-NEXT:  Classifying expressions for: @zext.nw.pre.inc
+; CHECK-NEXT:    %i = phi i8 [ -1, %entry ], [ %i.inc, %loop ]
+; CHECK-NEXT:    --> {-1,+,-128}<%loop> U: [-1,-128) S: [-1,-128) Exits: 127 LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %counter = phi i8 [ 0, %entry ], [ %counter.inc, %loop ]
+; CHECK-NEXT:    --> {0,+,1}<nuw><nsw><%loop> U: [0,2) S: [0,2) Exits: 1 LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %i.inc = add i8 %i, -128
+; CHECK-NEXT:    --> {127,+,-128}<%loop> U: [127,0) S: [127,0) Exits: -1 LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %i.sext = zext i8 %i to i16
+; CHECK-NEXT:    --> {255,+,-128}<nw><%loop> U: [127,256) S: [127,256) Exits: 127 LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %counter.inc = add i8 %counter, 1
+; CHECK-NEXT:    --> {1,+,1}<nuw><nsw><%loop> U: [1,3) S: [1,3) Exits: 2 LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:  Determining loop execution counts for: @zext.nw.pre.inc
+; CHECK-NEXT:  Loop %loop: backedge-taken count is i8 1
+; CHECK-NEXT:  Loop %loop: constant max backedge-taken count is i8 1
+; CHECK-NEXT:  Loop %loop: symbolic max backedge-taken count is i8 1
+; CHECK-NEXT:  Loop %loop: Trip multiple is 2
+;
+ entry:
+  br label %loop
+
+ loop:
+  %i = phi i8 [ -1, %entry ], [ %i.inc, %loop ]
+  %counter = phi i8 [ 0, %entry ], [ %counter.inc, %loop ]
+  %i.inc = add i8 %i, -128
+  %i.sext = zext i8 %i to i16
+  %counter.inc = add i8 %counter, 1
+  %continue = icmp eq i8 %counter, 1
+  br i1 %continue, label %exit, label %loop
+
+ exit:
+  ret void
+}
+
+; nw on the pre-inc does not imply nsuw on the sext.
+; The sext-addrec should not be folded.
+; The pre-inc AR {-1,+,-2} is nw-only: nw is inferred from the sgt-exit of a
+; finite loop whose step is a power of two. The constant 1 is first peeled off
+; the sext operand, so the varying-start argument runs for Delta = 1, between
+; {0,+,-2} and the pre-inc AR. nw is not enough: it only forbids the pre-inc
+; AR from revisiting its start, not from crossing the signed boundary. If %i
+; runs down through -2147483647 and crosses to 2147483647, 2147483645, then
+; %j takes the values 1, -1, ..., -2147483645, -2147483647, 2147483647. The
+; transition -2147483647 -> 2147483647 wraps past signed-min, so at that
+; iteration sext(%j) is 2147483647 whereas {1,+,-2} evaluated in i64 gives
+; -2147483649.
+define void @sext.nsuw.nw.pre.inc(ptr %buf, i32 %n) mustprogress {
+; CHECK-LABEL: 'sext.nsuw.nw.pre.inc'
+; CHECK-NEXT:  Classifying expressions for: @sext.nsuw.nw.pre.inc
+; CHECK-NEXT:    %i = phi i32 [ -1, %entry ], [ %i.next, %loop ]
+; CHECK-NEXT:    --> {-1,+,-2}<nw><%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %i.next = add i32 %i, -2
+; CHECK-NEXT:    --> {-3,+,-2}<nw><%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %j = add i32 %i, 2
+; CHECK-NEXT:    --> {1,+,-2}<nw><%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %idx = sext i32 %j to i64
+; CHECK-NEXT:    --> (1 + (sext i32 {0,+,-2}<nw><%loop> to i64))<nuw><nsw> U: [1,0) S: [-2147483647,2147483648) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %gep = getelementptr inbounds i64, ptr %buf, i64 %idx
+; CHECK-NEXT:    --> (8 + (8 * (sext i32 {0,+,-2}<nw><%loop> to i64))<nsw> + %buf) U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:  Determining loop execution counts for: @sext.nsuw.nw.pre.inc
+; CHECK-NEXT:  Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT:  Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT:  Loop %loop: Unpredictable symbolic max backedge-taken count.
+;
+ entry:
+  br label %loop
+
+ loop:
+  %i = phi i32 [ -1, %entry ], [ %i.next, %loop ]
+  %i.next = add i32 %i, -2
+  %j = add i32 %i, 2
+  %idx = sext i32 %j to i64
+  %gep = getelementptr inbounds i64, ptr %buf, i64 %idx
+  store i64 7, ptr %gep
+  %cmp = icmp sgt i32 %i, %n
+  br i1 %cmp, label %loop, label %exit
+
+ exit:
+  ret void
+}
+
+; nw on the pre-inc does not imply nusw on the zext.
+; The zext should not be folded.
+; The pre-inc AR {98,+,(-1 * vscale)} is nw-only: nw is inferred from the
+; eq-exit of a finite loop whose step is a power of two. The assume guarantees
+; %i.next u< 254 on every backedge, so all values of the pre-inc AR are
+; u< 255, which "proves" the no-overflow side of the varying-start argument
+; for Delta = 1. That is not enough: with vscale = 8 and n = 242, %i takes
+; the values 98, 90, ..., 2, 250, 242 (the assume holds throughout, and the
+; pre-inc AR does not self-wrap), while %j takes the values 99, 91, ..., 3,
+; 251, 243. The transition 3 -> 251 wraps below zero, so at that iteration
+; zext(%j) is 251 whereas {99,+,-8} evaluated in i16 gives -5.
+define void @zext.nusw.nw.pre.inc(i8 %n) mustprogress willreturn {
+; CHECK-LABEL: 'zext.nusw.nw.pre.inc'
+; CHECK-NEXT:  Classifying expressions for: @zext.nusw.nw.pre.inc
+; CHECK-NEXT:    %vs = call i8 @llvm.vscale.i8()
+; CHECK-NEXT:    --> vscale U: [1,0) S: [1,0)
+; CHECK-NEXT:    %step = sub i8 0, %vs
+; CHECK-NEXT:    --> (-1 * vscale) U: [1,0) S: [1,0)
+; CHECK-NEXT:    %i = phi i8 [ 98, %entry ], [ %i.next, %loop ]
+; CHECK-NEXT:    --> {98,+,(-1 * vscale)}<nw><%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %i.next = add i8 %i, %step
+; CHECK-NEXT:    --> {(98 + (-1 * vscale)),+,(-1 * vscale)}<nw><%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %j = add i8 %i, 1
+; CHECK-NEXT:    --> {99,+,(-1 * vscale)}<nw><%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    %idx = zext i8 %j to i16
+; CHECK-NEXT:    --> (zext i8 {99,+,(-1 * vscale)}<nw><%loop> to i16) U: [0,256) S: [0,256) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:  Determining loop execution counts for: @zext.nusw.nw.pre.inc
+; CHECK-NEXT:  Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT:  Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT:  Loop %loop: Unpredictable symbolic max backedge-taken count.
+;
+entry:
+  %vs = call i8 @llvm.vscale.i8()
+  %step = sub i8 0, %vs
+  br label %loop
+
+loop:
+  %i = phi i8 [ 98, %entry ], [ %i.next, %loop ]
+  %i.next = add i8 %i, %step
+  %j = add i8 %i, 1
+  %idx = zext i8 %j to i16
+  %in.bounds = icmp ult i8 %i.next, -2
+  call void @llvm.assume(i1 %in.bounds)
+  %exitcond = icmp eq i8 %i, %n
+  br i1 %exitcond, label %exit, label %loop
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Analysis/ScalarEvolution/incorrect-nsw.ll b/llvm/test/Analysis/ScalarEvolution/incorrect-nsw.ll
deleted file mode 100644
index f7edf493c64c4..0000000000000
--- a/llvm/test/Analysis/ScalarEvolution/incorrect-nsw.ll
+++ /dev/null
@@ -1,26 +0,0 @@
-; RUN: opt -disable-output "-passes=print<scalar-evolution>,print<scalar-evolution>" < %s 2>&1 | FileCheck %s
-
-define void @bad.nsw() {
-; CHECK-LABEL: Classifying expressions for: @bad.nsw
-; CHECK-LABEL: Classifying expressions for: @bad.nsw
- entry: 
-  br label %loop
-
- loop:
-  %i = phi i8 [ -1, %entry ], [ %i.inc, %loop ]
-; CHECK:  %i = phi i8 [ -1, %entry ], [ %i.inc, %loop ]
-; CHECK-NEXT: -->  {-1,+,-128}<nw><%loop>
-; CHECK-NOT: -->  {-1,+,-128}<nsw><%loop>
-
-  %counter = phi i8 [ 0, %entry ], [ %counter.inc, %loop ]
-
-  %i.inc = add i8 %i, -128
-  %i.sext = sext i8 %i to i16
-
-  %counter.inc = add i8 %counter, 1
-  %continue = icmp eq i8 %counter, 1
-  br i1 %continue, label %exit, label %loop
-
- exit:
-  ret void  
-}

>From e44cbde87eefb6cea39ba092016f017e9a4c72ef Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Wed, 19 Aug 2026 18:36:12 +0100
Subject: [PATCH 2/5] [SCEV] Rework wrap-flag-inferrence for sext-addrec

Rewrite getSignExtendExprImpl to avoid the roundabout method of creating
sign-extend expressions to check no-wrap, by computing the no-wrap
information using induction and reading it off the expression directly.
The new code has much better compile-time, while not being exactly
equivalent to the old code.
---
 llvm/lib/Analysis/ScalarEvolution.cpp         | 95 ++-----------------
 .../ScalarEvolution/ext-addrec-wrap-flags.ll  |  2 +-
 2 files changed, 8 insertions(+), 89 deletions(-)

diff --git a/llvm/lib/Analysis/ScalarEvolution.cpp b/llvm/lib/Analysis/ScalarEvolution.cpp
index 6b0951491a88a..0d045a384431a 100644
--- a/llvm/lib/Analysis/ScalarEvolution.cpp
+++ b/llvm/lib/Analysis/ScalarEvolution.cpp
@@ -1994,91 +1994,17 @@ const SCEV *ScalarEvolution::getSignExtendExprImpl(SCEVUse Op, Type *Ty,
   // operands (often constants).  This allows analysis of something like
   // this:  for (signed char X = 0; X < 100; ++X) { int Y = X; }
   if (match(Op, m_scev_AffineAddRec(m_SCEV(Start), m_SCEV(Step), m_Loop(L)))) {
+    // Redo the AddRec check, computing nsw this time.
     const auto *AR = cast<SCEVAddRecExpr>(Op);
-    unsigned BitWidth = getTypeSizeInBits(AR->getType());
-
-    // The no-signed-wrap case is handled before the uniquing lookup above.
-
-    // Check whether the backedge-taken count is SCEVCouldNotCompute.
-    // Note that this serves two purposes: It filters out loops that are
-    // simply not analyzable, and it covers the case where this code is
-    // being called from within backedge-taken count analysis, such that
-    // attempting to ask for the backedge-taken count would likely result
-    // in infinite recursion. In the later case, the analysis code will
-    // cope with a conservative value, and it will take care to purge
-    // that value once it has finished.
-    const SCEV *MaxBECount = getConstantMaxBackedgeTakenCount(L);
-    if (!isa<SCEVCouldNotCompute>(MaxBECount)) {
-      // Manually compute the final value for AR, checking for
-      // overflow.
-
-      // Check whether the backedge-taken count can be losslessly casted to
-      // the addrec's type. The count is always unsigned.
-      const SCEV *CastedMaxBECount =
-          getTruncateOrZeroExtend(MaxBECount, Start->getType(), Depth);
-      const SCEV *RecastedMaxBECount = getTruncateOrZeroExtend(
-          CastedMaxBECount, MaxBECount->getType(), Depth);
-      if (MaxBECount == RecastedMaxBECount) {
-        Type *WideTy = IntegerType::get(getContext(), BitWidth * 2);
-        // Check whether Start+Step*MaxBECount has no signed overflow.
-        const SCEV *SMul =
-            getMulExpr(CastedMaxBECount, Step, SCEV::FlagAnyWrap, Depth + 1);
-        const SCEV *SAdd = getSignExtendExpr(
-            getAddExpr(Start, SMul, SCEV::FlagAnyWrap, Depth + 1), WideTy,
-            Depth + 1);
-        const SCEV *WideStart = getSignExtendExpr(Start, WideTy, Depth + 1);
-        const SCEV *WideMaxBECount =
-            getZeroExtendExpr(CastedMaxBECount, WideTy, Depth + 1);
-        const SCEV *OperandExtendedAdd =
-            getAddExpr(WideStart,
-                       getMulExpr(WideMaxBECount,
-                                  getSignExtendExpr(Step, WideTy, Depth + 1),
-                                  SCEV::FlagAnyWrap, Depth + 1),
-                       SCEV::FlagAnyWrap, Depth + 1);
-        if (SAdd == OperandExtendedAdd) {
-          // Cache knowledge of AR NSW, which is propagated to this AddRec.
-          setNoWrapFlags(const_cast<SCEVAddRecExpr *>(AR), SCEV::FlagNSW);
-          // Return the expression with the addrec on the outside.
-          Start =
-              getExtendAddRecStart<SCEVSignExtendExpr>(AR, Ty, this, Depth + 1);
-          Step = getSignExtendExpr(Step, Ty, Depth + 1);
-          return getAddRecExpr(Start, Step, L, AR->getNoWrapFlags());
-        }
-        // Similar to above, only this time treat the step value as unsigned.
-        // This covers loops that count up with an unsigned step.
-        OperandExtendedAdd =
-            getAddExpr(WideStart,
-                       getMulExpr(WideMaxBECount,
-                                  getZeroExtendExpr(Step, WideTy, Depth + 1),
-                                  SCEV::FlagAnyWrap, Depth + 1),
-                       SCEV::FlagAnyWrap, Depth + 1);
-        if (SAdd == OperandExtendedAdd) {
-          // If AR wraps around then
-          //
-          //    abs(Step) * MaxBECount > unsigned-max(AR->getType())
-          // => SAdd != OperandExtendedAdd
-          //
-          // Thus (AR is not NW => SAdd != OperandExtendedAdd) <=>
-          // (SAdd == OperandExtendedAdd => AR is NW)
-
-          setNoWrapFlags(const_cast<SCEVAddRecExpr *>(AR), SCEV::FlagNW);
-
-          // Return the expression with the addrec on the outside.
-          Start =
-              getExtendAddRecStart<SCEVSignExtendExpr>(AR, Ty, this, Depth + 1);
-          Step = getZeroExtendExpr(Step, Ty, Depth + 1);
-          return getAddRecExpr(Start, Step, L, AR->getNoWrapFlags());
-        }
-      }
-    }
-
+    inferNoWrapViaConstantRanges(AR);
     auto NewFlags = proveNoSignedWrapViaInduction(AR);
+    if (!hasFlags(NewFlags, SCEV::FlagNSW) &&
+        proveNoWrapByVaryingStart<SCEVSignExtendExpr>(Start, Step, L))
+      NewFlags |= SCEV::FlagNSW;
     setNoWrapFlags(const_cast<SCEVAddRecExpr *>(AR), NewFlags);
+
+    // If we have nsw, the sign-extend distributes over the recurrence.
     if (AR->hasNoSignedWrap()) {
-      // Same as nsw case above - duplicated here to avoid a compile time
-      // issue.  It's not clear that the order of checks does matter, but
-      // it's one of two issue possible causes for a change which was
-      // reverted.  Be conservative for the moment.
       Start = getExtendAddRecStart<SCEVSignExtendExpr>(AR, Ty, this, Depth + 1);
       Step = getSignExtendExpr(Step, Ty, Depth + 1);
       return getAddRecExpr(Start, Step, L, AR->getNoWrapFlags());
@@ -2099,13 +2025,6 @@ const SCEV *ScalarEvolution::getSignExtendExprImpl(SCEVUse Op, Type *Ty,
                           Depth + 1);
       }
     }
-
-    if (proveNoWrapByVaryingStart<SCEVSignExtendExpr>(Start, Step, L)) {
-      setNoWrapFlags(const_cast<SCEVAddRecExpr *>(AR), SCEV::FlagNSW);
-      Start = getExtendAddRecStart<SCEVSignExtendExpr>(AR, Ty, this, Depth + 1);
-      Step = getSignExtendExpr(Step, Ty, Depth + 1);
-      return getAddRecExpr(Start, Step, L, AR->getNoWrapFlags());
-    }
   }
 
   // If the input value is provably positive and we could not simplify
diff --git a/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll b/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
index 9e1e1387eead5..071ddec4bcd69 100644
--- a/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
+++ b/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
@@ -13,7 +13,7 @@ define void @sext.nw.pre.inc() {
 ; CHECK-NEXT:    %i.inc = add i8 %i, -128
 ; CHECK-NEXT:    --> {127,+,-128}<%loop> U: [127,0) S: [127,0) Exits: -1 LoopDispositions: { %loop: Computable }
 ; CHECK-NEXT:    %i.sext = sext i8 %i to i16
-; CHECK-NEXT:    --> {-1,+,128}<nw><%loop> U: [-1,128) S: [-1,128) Exits: 127 LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    --> (127 + (sext i8 {-128,+,-128}<%loop> to i16))<nuw><nsw> U: [127,0) S: [-1,128) Exits: 127 LoopDispositions: { %loop: Computable }
 ; CHECK-NEXT:    %counter.inc = add i8 %counter, 1
 ; CHECK-NEXT:    --> {1,+,1}<nuw><nsw><%loop> U: [1,3) S: [1,3) Exits: 2 LoopDispositions: { %loop: Computable }
 ; CHECK-NEXT:  Determining loop execution counts for: @sext.nw.pre.inc

>From 705f99410441bff3304d0c5affb3dbbf24a83261 Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Fri, 21 Aug 2026 20:29:25 +0100
Subject: [PATCH 3/5] [SCEV] Cover old no-self-wrap logic

---
 llvm/include/llvm/Analysis/ScalarEvolution.h |  2 +-
 llvm/lib/Analysis/ScalarEvolution.cpp        | 23 +++++++++++++++-----
 2 files changed, 18 insertions(+), 7 deletions(-)

diff --git a/llvm/include/llvm/Analysis/ScalarEvolution.h b/llvm/include/llvm/Analysis/ScalarEvolution.h
index b162036ffb46d..61570a8549ad8 100644
--- a/llvm/include/llvm/Analysis/ScalarEvolution.h
+++ b/llvm/include/llvm/Analysis/ScalarEvolution.h
@@ -2417,7 +2417,7 @@ class ScalarEvolution {
   /// equivalent to proving no signed (resp. unsigned) wrap in
   /// {`Start`,+,`Step`} if `ExtendOpTy` is `SCEVSignExtendExpr`
   /// (resp. `SCEVZeroExtendExpr`).
-  template <typename ExtendOpTy>
+  template <typename ExtendOpTy, SCEVNoWrapFlags WrapType>
   bool proveNoWrapByVaryingStart(const SCEV *Start, const SCEV *Step,
                                  const Loop *L);
 
diff --git a/llvm/lib/Analysis/ScalarEvolution.cpp b/llvm/lib/Analysis/ScalarEvolution.cpp
index 0d045a384431a..d609984054399 100644
--- a/llvm/lib/Analysis/ScalarEvolution.cpp
+++ b/llvm/lib/Analysis/ScalarEvolution.cpp
@@ -1421,12 +1421,10 @@ static const SCEV *getExtendAddRecStart(const SCEVAddRecExpr *AR, Type *Ty,
 //
 // In the current context, S is `Start`, X is `Step`, Ext is `ExtendOpTy` and T
 // is `Delta` (defined below).
-template <typename ExtendOpTy>
+template <typename ExtendOpTy, SCEVNoWrapFlags WrapType>
 bool ScalarEvolution::proveNoWrapByVaryingStart(const SCEV *Start,
                                                 const SCEV *Step,
                                                 const Loop *L) {
-  auto WrapType = ExtendOpTraits<ExtendOpTy>::WrapType;
-
   // We restrict `Start` to a constant to prevent SCEV from spending too much
   // time here.  It is correct (but more expensive) to continue with a
   // non-constant `Start` and do a general SCEV subtraction to compute
@@ -1735,7 +1733,8 @@ const SCEV *ScalarEvolution::getZeroExtendExprImpl(SCEVUse Op, Type *Ty,
       }
     }
 
-    if (proveNoWrapByVaryingStart<SCEVZeroExtendExpr>(Start, Step, L)) {
+    if (proveNoWrapByVaryingStart<SCEVZeroExtendExpr, SCEV::FlagNUW>(Start,
+                                                                     Step, L)) {
       setNoWrapFlags(const_cast<SCEVAddRecExpr *>(AR), SCEV::FlagNUW);
       Start = getExtendAddRecStart<SCEVZeroExtendExpr>(AR, Ty, this, Depth + 1);
       Step = getZeroExtendExpr(Step, Ty, Depth + 1);
@@ -1994,12 +1993,13 @@ const SCEV *ScalarEvolution::getSignExtendExprImpl(SCEVUse Op, Type *Ty,
   // operands (often constants).  This allows analysis of something like
   // this:  for (signed char X = 0; X < 100; ++X) { int Y = X; }
   if (match(Op, m_scev_AffineAddRec(m_SCEV(Start), m_SCEV(Step), m_Loop(L)))) {
-    // Redo the AddRec check, computing nsw this time.
+    // Redo the AddRec check, attempting to prove no-wrap this time.
     const auto *AR = cast<SCEVAddRecExpr>(Op);
     inferNoWrapViaConstantRanges(AR);
     auto NewFlags = proveNoSignedWrapViaInduction(AR);
     if (!hasFlags(NewFlags, SCEV::FlagNSW) &&
-        proveNoWrapByVaryingStart<SCEVSignExtendExpr>(Start, Step, L))
+        proveNoWrapByVaryingStart<SCEVSignExtendExpr, SCEV::FlagNSW>(Start,
+                                                                     Step, L))
       NewFlags |= SCEV::FlagNSW;
     setNoWrapFlags(const_cast<SCEVAddRecExpr *>(AR), NewFlags);
 
@@ -2025,6 +2025,17 @@ const SCEV *ScalarEvolution::getSignExtendExprImpl(SCEVUse Op, Type *Ty,
                           Depth + 1);
       }
     }
+
+    // Handle the nsuw case. nuw on pre-inc AR and no-overflow proven with
+    // signed overflow limit implies nsuw on the parent AR, of which nw is a
+    // weaker version.
+    if (proveNoWrapByVaryingStart<SCEVSignExtendExpr, SCEV::FlagNUW>(Start,
+                                                                     Step, L)) {
+      setNoWrapFlags(const_cast<SCEVAddRecExpr *>(AR), SCEV::FlagNW);
+      Start = getExtendAddRecStart<SCEVSignExtendExpr>(AR, Ty, this, Depth + 1);
+      Step = getZeroExtendExpr(Step, Ty, Depth + 1);
+      return getAddRecExpr(Start, Step, L, AR->getNoWrapFlags());
+    }
   }
 
   // If the input value is provably positive and we could not simplify

>From 29e29743c9eea36008987e56541f3f9dc088106e Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Sun, 23 Aug 2026 09:42:08 +0100
Subject: [PATCH 4/5] [SCEV] Is nw sufficient to prove nsuw?

---
 llvm/lib/Analysis/ScalarEvolution.cpp                  | 10 +++++-----
 .../Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll  |  4 ++--
 2 files changed, 7 insertions(+), 7 deletions(-)

diff --git a/llvm/lib/Analysis/ScalarEvolution.cpp b/llvm/lib/Analysis/ScalarEvolution.cpp
index d609984054399..fcf4a6ac931de 100644
--- a/llvm/lib/Analysis/ScalarEvolution.cpp
+++ b/llvm/lib/Analysis/ScalarEvolution.cpp
@@ -2026,11 +2026,11 @@ const SCEV *ScalarEvolution::getSignExtendExprImpl(SCEVUse Op, Type *Ty,
       }
     }
 
-    // Handle the nsuw case. nuw on pre-inc AR and no-overflow proven with
-    // signed overflow limit implies nsuw on the parent AR, of which nw is a
-    // weaker version.
-    if (proveNoWrapByVaryingStart<SCEVSignExtendExpr, SCEV::FlagNUW>(Start,
-                                                                     Step, L)) {
+    // Handle the nsuw case. nw on pre-inc AR and no-overflow proven with signed
+    // overflow limit implies nsuw on the parent AR, of which nw is a weaker
+    // version.
+    if (proveNoWrapByVaryingStart<SCEVSignExtendExpr, SCEV::FlagNW>(Start, Step,
+                                                                    L)) {
       setNoWrapFlags(const_cast<SCEVAddRecExpr *>(AR), SCEV::FlagNW);
       Start = getExtendAddRecStart<SCEVSignExtendExpr>(AR, Ty, this, Depth + 1);
       Step = getZeroExtendExpr(Step, Ty, Depth + 1);
diff --git a/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll b/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
index 071ddec4bcd69..dcff3ce7e3f84 100644
--- a/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
+++ b/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
@@ -98,9 +98,9 @@ define void @sext.nsuw.nw.pre.inc(ptr %buf, i32 %n) mustprogress {
 ; CHECK-NEXT:    %j = add i32 %i, 2
 ; CHECK-NEXT:    --> {1,+,-2}<nw><%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
 ; CHECK-NEXT:    %idx = sext i32 %j to i64
-; CHECK-NEXT:    --> (1 + (sext i32 {0,+,-2}<nw><%loop> to i64))<nuw><nsw> U: [1,0) S: [-2147483647,2147483648) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    --> {1,+,4294967294}<nuw><%loop> U: [1,0) S: [1,0) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
 ; CHECK-NEXT:    %gep = getelementptr inbounds i64, ptr %buf, i64 %idx
-; CHECK-NEXT:    --> (8 + (8 * (sext i32 {0,+,-2}<nw><%loop> to i64))<nsw> + %buf) U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    --> {(8 + %buf),+,34359738352}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
 ; CHECK-NEXT:  Determining loop execution counts for: @sext.nsuw.nw.pre.inc
 ; CHECK-NEXT:  Loop %loop: Unpredictable backedge-taken count.
 ; CHECK-NEXT:  Loop %loop: Unpredictable constant max backedge-taken count.

>From ee4fa129021ff3475f83b1d7800c318656110a94 Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Wed, 26 Aug 2026 15:33:24 +0100
Subject: [PATCH 5/5] Revert "[SCEV] Is nw sufficient to prove nsuw?"

This reverts commit dd5c30a4de034ce107d8b9dafdaaac4dee70b9bf.
---
 llvm/lib/Analysis/ScalarEvolution.cpp                  | 10 +++++-----
 .../Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll  |  4 ++--
 2 files changed, 7 insertions(+), 7 deletions(-)

diff --git a/llvm/lib/Analysis/ScalarEvolution.cpp b/llvm/lib/Analysis/ScalarEvolution.cpp
index fcf4a6ac931de..d609984054399 100644
--- a/llvm/lib/Analysis/ScalarEvolution.cpp
+++ b/llvm/lib/Analysis/ScalarEvolution.cpp
@@ -2026,11 +2026,11 @@ const SCEV *ScalarEvolution::getSignExtendExprImpl(SCEVUse Op, Type *Ty,
       }
     }
 
-    // Handle the nsuw case. nw on pre-inc AR and no-overflow proven with signed
-    // overflow limit implies nsuw on the parent AR, of which nw is a weaker
-    // version.
-    if (proveNoWrapByVaryingStart<SCEVSignExtendExpr, SCEV::FlagNW>(Start, Step,
-                                                                    L)) {
+    // Handle the nsuw case. nuw on pre-inc AR and no-overflow proven with
+    // signed overflow limit implies nsuw on the parent AR, of which nw is a
+    // weaker version.
+    if (proveNoWrapByVaryingStart<SCEVSignExtendExpr, SCEV::FlagNUW>(Start,
+                                                                     Step, L)) {
       setNoWrapFlags(const_cast<SCEVAddRecExpr *>(AR), SCEV::FlagNW);
       Start = getExtendAddRecStart<SCEVSignExtendExpr>(AR, Ty, this, Depth + 1);
       Step = getZeroExtendExpr(Step, Ty, Depth + 1);
diff --git a/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll b/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
index dcff3ce7e3f84..071ddec4bcd69 100644
--- a/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
+++ b/llvm/test/Analysis/ScalarEvolution/ext-addrec-wrap-flags.ll
@@ -98,9 +98,9 @@ define void @sext.nsuw.nw.pre.inc(ptr %buf, i32 %n) mustprogress {
 ; CHECK-NEXT:    %j = add i32 %i, 2
 ; CHECK-NEXT:    --> {1,+,-2}<nw><%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
 ; CHECK-NEXT:    %idx = sext i32 %j to i64
-; CHECK-NEXT:    --> {1,+,4294967294}<nuw><%loop> U: [1,0) S: [1,0) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    --> (1 + (sext i32 {0,+,-2}<nw><%loop> to i64))<nuw><nsw> U: [1,0) S: [-2147483647,2147483648) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
 ; CHECK-NEXT:    %gep = getelementptr inbounds i64, ptr %buf, i64 %idx
-; CHECK-NEXT:    --> {(8 + %buf),+,34359738352}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT:    --> (8 + (8 * (sext i32 {0,+,-2}<nw><%loop> to i64))<nsw> + %buf) U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
 ; CHECK-NEXT:  Determining loop execution counts for: @sext.nsuw.nw.pre.inc
 ; CHECK-NEXT:  Loop %loop: Unpredictable backedge-taken count.
 ; CHECK-NEXT:  Loop %loop: Unpredictable constant max backedge-taken count.



More information about the llvm-commits mailing list