[llvm] [SCEVExpander] Reuse unscaled recurrence when expanding scaled one. (PR #223850)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 22 02:30:19 PDT 2026


https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/223850

>From 12187b99acf258a61d5dafc4022c1071da7d19ac Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 15 Sep 2026 22:27:12 +0100
Subject: [PATCH 1/2] Precommit test

---
 .../runtime-checks-difference.ll              | 113 ++++++++++++++++++
 1 file changed, 113 insertions(+)

diff --git a/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll b/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
index e816b559bffe6..dc7feef799b80 100644
--- a/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
+++ b/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
@@ -566,6 +566,119 @@ exit:
   ret void
 }
 
+; Same as @nested_loop_bound_needs_constant_correction, but the loop already
+; has an induction variable stepping by 8. The bound is expanded as an offset
+; from it.
+define void @nested_loop_bound_reuses_scaled_iv(ptr %a, ptr %b, ptr %c, i64 %n) {
+; CHECK-LABEL: define void @nested_loop_bound_reuses_scaled_iv(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = shl i64 [[N]], 2
+; CHECK-NEXT:    [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP1:%.*]] = shl i64 [[N]], 3
+; CHECK-NEXT:    [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH:%.*]] ]
+; CHECK-NEXT:    [[SCALED_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[SCALED_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = shl i64 [[OUTER_IV]], 2
+; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP3:%.*]] = add i64 [[SCALED_IV]], 4
+; CHECK-NEXT:    [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT:    [[TMP4:%.*]] = sub i64 [[N]], [[OUTER_IV]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP4]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_MEMCHECK:.*]]
+; CHECK:       [[VECTOR_MEMCHECK]]:
+; CHECK-NEXT:    [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP3]]
+; CHECK-NEXT:    [[BOUND1:%.*]] = icmp ult ptr [[SCEVGEP2]], [[SCEVGEP1]]
+; CHECK-NEXT:    [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
+; CHECK-NEXT:    br i1 [[FOUND_CONFLICT]], [[SCALAR_PH]], [[VECTOR_PH:label %.*]]
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+  %scaled.iv = phi i64 [ 0, %entry ], [ %scaled.iv.next, %outer.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ %outer.iv, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %gep.a = getelementptr inbounds { i32, i32 }, ptr %a, i64 %inner.iv, i32 1
+  %l = load i32, ptr %gep.a, align 4
+  %gep.b = getelementptr inbounds i32, ptr %b, i64 %inner.iv
+  store i32 %l, ptr %gep.b, align 4
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.cond = icmp eq i64 %inner.iv.next, %n
+  br i1 %inner.cond, label %outer.latch, label %inner.body
+
+outer.latch:
+  %gep.c = getelementptr inbounds i8, ptr %c, i64 %scaled.iv
+  store i64 0, ptr %gep.c, align 8
+  %scaled.iv.next = add nuw nsw i64 %scaled.iv, 8
+  %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+  %outer.cond = icmp eq i64 %outer.iv.next, %n
+  br i1 %outer.cond, label %exit, label %outer.header
+
+exit:
+  ret void
+}
+
+; Same as @nested_loop_bound_needs_constant_correction, but with an outer
+; induction variable counting down. The bounds are expanded by scaling the IV.
+define void @nested_loop_bound_with_negative_step(ptr %a, ptr %b, i64 %n) {
+; CHECK-LABEL: define void @nested_loop_bound_with_negative_step(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = shl i64 [[N]], 2
+; CHECK-NEXT:    [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP1:%.*]] = shl i64 [[N]], 3
+; CHECK-NEXT:    [[TMP6:%.*]] = add nuw nsw i64 [[TMP1]], 4
+; CHECK-NEXT:    [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[INDVAR:%.*]] = phi i64 [ [[INDVAR_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ [[N]], %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
+; CHECK-NEXT:    [[TMP3:%.*]] = mul i64 [[INDVAR]], -4
+; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[TMP0]], [[TMP3]]
+; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP5:%.*]] = mul i64 [[INDVAR]], -8
+; CHECK-NEXT:    [[TMP4:%.*]] = add i64 [[TMP6]], [[TMP5]]
+; CHECK-NEXT:    [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[INDVAR]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_MEMCHECK:.*]]
+; CHECK:       [[VECTOR_MEMCHECK]]:
+; CHECK-NEXT:    [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP3]]
+; CHECK-NEXT:    [[BOUND1:%.*]] = icmp ult ptr [[SCEVGEP2]], [[SCEVGEP1]]
+; CHECK-NEXT:    [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
+; CHECK-NEXT:    br i1 [[FOUND_CONFLICT]], [[SCALAR_PH]], [[VECTOR_PH:label %.*]]
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %outer.iv = phi i64 [ %n, %entry ], [ %outer.iv.next, %outer.latch ]
+  br label %inner.body
+
+inner.body:
+  %inner.iv = phi i64 [ %outer.iv, %outer.header ], [ %inner.iv.next, %inner.body ]
+  %gep.a = getelementptr inbounds { i32, i32 }, ptr %a, i64 %inner.iv, i32 1
+  %l = load i32, ptr %gep.a, align 4
+  %gep.b = getelementptr inbounds i32, ptr %b, i64 %inner.iv
+  store i32 %l, ptr %gep.b, align 4
+  %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+  %inner.cond = icmp eq i64 %inner.iv.next, %n
+  br i1 %inner.cond, label %outer.latch, label %inner.body
+
+outer.latch:
+  %outer.iv.next = add nsw i64 %outer.iv, -1
+  %outer.cond = icmp eq i64 %outer.iv.next, 0
+  br i1 %outer.cond, label %exit, label %outer.header
+
+exit:
+  ret void
+}
+
 ; Test case where the AddRec for the pointers in the inner loop have the AddRec
 ; of the outer loop as start value. It is sufficient to subtract the start
 ; values (%dst, %src) of the outer AddRecs.

>From 5ba772b37715b93e1713cf70f6cd1d589dde1b06 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 19 Aug 2026 13:16:46 +0100
Subject: [PATCH 2/2] [SCEVExpander] Reuse unscaled recurrence when expanding
 scaled one.

When expanding a scaled recurrence, like {8*X,+,8}, we can re-use an
existing {X,+,1} recurrence and just scale the existing result. This
avoids introducing new IVs in loops unnecessarily.

This improves SCEV expansions in a number of cases:
https://github.com/dtcxzyw/llvm-opt-benchmark-nightly/pull/1320.

Found while investigating small SCEV expansion regressions due to
additional folding/flag inference in ConstraintElimination.
---
 .../Utils/ScalarEvolutionExpander.h           |  1 +
 .../Utils/ScalarEvolutionExpander.cpp         | 55 +++++++++++++++++++
 .../AMDGPU/vmem-cache-line-size.ll            | 20 +++----
 .../reuse-lcssa-phi-scev-expansion.ll         |  3 +-
 .../LoopVectorize/VPlan/memory-checks.ll      |  2 +-
 .../runtime-checks-difference.ll              | 52 ++++++++----------
 6 files changed, 91 insertions(+), 42 deletions(-)

diff --git a/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h b/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h
index c98c0cb52fa9c..6886342721399 100644
--- a/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h
+++ b/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h
@@ -561,6 +561,7 @@ class SCEVExpander : public SCEVUseVisitor<SCEVExpander, Value *> {
   bool isExpandedAddRecExprPHI(PHINode *PN, Instruction *IncV, const Loop *L);
 
   Value *tryToReuseLCSSAPhi(SCEVUseT<const SCEVAddRecExpr *> S);
+  Value *tryToReuseScaledAddRec(const SCEVAddRecExpr *S);
   Value *expandAddRecExprLiterally(SCEVUseT<const SCEVAddRecExpr *> S);
   PHINode *getAddRecExprPHILiterally(const SCEVAddRecExpr *Normalized,
                                      const Loop *L, Type *&TruncTy,
diff --git a/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp b/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
index 93baf01173f98..3d3de1ff7020c 100644
--- a/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
+++ b/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
@@ -17,6 +17,7 @@
 #include "llvm/ADT/ScopeExit.h"
 #include "llvm/Analysis/InstructionSimplify.h"
 #include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/ScalarEvolutionDivision.h"
 #include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
 #include "llvm/Analysis/TargetTransformInfo.h"
 #include "llvm/Analysis/ValueTracking.h"
@@ -1334,6 +1335,57 @@ Value *SCEVExpander::tryToReuseLCSSAPhi(SCEVUseT<const SCEVAddRecExpr *> S) {
   return nullptr;
 }
 
+/// Try to re-use an existing AddRec when expanding S, if
+/// AddRec * Scale + Offset == S for constant Scale and Offset.
+Value *SCEVExpander::tryToReuseScaledAddRec(const SCEVAddRecExpr *S) {
+  const APInt *Step;
+  if (!S->getType()->isIntegerTy() ||
+      !match(S->getStepRecurrence(SE), m_scev_APInt(Step)))
+    return nullptr;
+
+  APInt Factor = Step->abs();
+  if (Factor.ule(1))
+    return nullptr;
+
+  const SCEV *FactorS = SE.getConstant(Factor);
+
+  // Split S into Divided * FactorS + Offset, with a constant Offset, and check
+  // if there is an existing Divided that can be scaled back to S.
+  const SCEV *Divided, *Offset;
+  SCEVDivision::divide(SE, S, FactorS, &Divided, &Offset);
+  if (!isa<SCEVConstant>(Offset))
+    return nullptr;
+  const SCEV *Scaled = SE.getMulExpr(Divided, FactorS);
+  if (SE.getAddExpr(Scaled, Offset) != S)
+    return nullptr;
+
+  // First check if we have an existing expansion for Scaled directly. Otherwise
+  // look for Divided and scale manually.
+  const Instruction *InsertPt = &*Builder.GetInsertPoint();
+  Value *V = nullptr;
+  if (!Offset->isZero())
+    V = findExistingExpansionAndDropPoisonFlags(Scaled, InsertPt);
+
+  if (!V) {
+    Value *Base = findExistingExpansionAndDropPoisonFlags(Divided, InsertPt);
+    if (!Base)
+      return nullptr;
+    // Carry over the no-wrap facts that hold for the scaling, otherwise the new
+    // expansion may miss flags the original one had.
+    SCEV::NoWrapFlags Flags = SCEV::FlagAnyWrap;
+    if (SE.willNotOverflow(Instruction::Mul, /*Signed=*/false, Divided,
+                           FactorS))
+      Flags = ScalarEvolution::setFlags(Flags, SCEV::FlagNUW);
+    if (SE.willNotOverflow(Instruction::Mul, /*Signed=*/true, Divided, FactorS))
+      Flags = ScalarEvolution::setFlags(Flags, SCEV::FlagNSW);
+    V = expand(SCEVUse(SE.getMulExpr(SE.getUnknown(Base), FactorS), Flags));
+  }
+
+  if (Offset->isZero())
+    return V;
+  return expand(SE.getAddExpr(SE.getUnknown(V), Offset));
+}
+
 Value *SCEVExpander::visitAddRecExpr(SCEVUseT<const SCEVAddRecExpr *> S) {
   // In canonical mode we compute the addrec as an expression of a canonical IV
   // using evaluateAtIteration and expand the resulting SCEV expression. This
@@ -1348,6 +1400,9 @@ Value *SCEVExpander::visitAddRecExpr(SCEVUseT<const SCEVAddRecExpr *> S) {
   if (!CanonicalMode || (S->getNumOperands() > 2))
     return expandAddRecExprLiterally(S);
 
+  if (Value *V = tryToReuseScaledAddRec(S))
+    return V;
+
   Type *Ty = SE.getEffectiveSCEVType(S->getType());
   const Loop *L = S->getLoop();
 
diff --git a/llvm/test/Transforms/LoopDataPrefetch/AMDGPU/vmem-cache-line-size.ll b/llvm/test/Transforms/LoopDataPrefetch/AMDGPU/vmem-cache-line-size.ll
index 8cca57c170eee..ba65b202387ec 100644
--- a/llvm/test/Transforms/LoopDataPrefetch/AMDGPU/vmem-cache-line-size.ll
+++ b/llvm/test/Transforms/LoopDataPrefetch/AMDGPU/vmem-cache-line-size.ll
@@ -10,7 +10,7 @@ define amdgpu_kernel void @prefetch_two_streams_80B(ptr addrspace(1) nocapture %
 ; GFX1250-NEXT:    br label %[[FOR_BODY:.*]]
 ; GFX1250:       [[FOR_BODY]]:
 ; GFX1250-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1250-NEXT:    [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1250-NEXT:    [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
 ; GFX1250-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 28
 ; GFX1250-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
 ; GFX1250-NEXT:    [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
@@ -34,7 +34,7 @@ define amdgpu_kernel void @prefetch_two_streams_80B(ptr addrspace(1) nocapture %
 ; GFX1200-NEXT:    br label %[[FOR_BODY:.*]]
 ; GFX1200:       [[FOR_BODY]]:
 ; GFX1200-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1200-NEXT:    [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1200-NEXT:    [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
 ; GFX1200-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 28
 ; GFX1200-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
 ; GFX1200-NEXT:    [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
@@ -100,10 +100,10 @@ define amdgpu_kernel void @prefetch_two_streams_160B(ptr addrspace(1) nocapture
 ; GFX1250-NEXT:    br label %[[FOR_BODY:.*]]
 ; GFX1250:       [[FOR_BODY]]:
 ; GFX1250-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1250-NEXT:    [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1250-NEXT:    [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
 ; GFX1250-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 188
 ; GFX1250-NEXT:    [[SCEVGEP1:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
-; GFX1250-NEXT:    [[TMP2:%.*]] = shl i64 [[IV]], 2
+; GFX1250-NEXT:    [[TMP2:%.*]] = shl nuw nsw i64 [[IV]], 2
 ; GFX1250-NEXT:    [[TMP3:%.*]] = add i64 [[TMP2]], 28
 ; GFX1250-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP3]]
 ; GFX1250-NEXT:    [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
@@ -128,10 +128,10 @@ define amdgpu_kernel void @prefetch_two_streams_160B(ptr addrspace(1) nocapture
 ; GFX1200-NEXT:    br label %[[FOR_BODY:.*]]
 ; GFX1200:       [[FOR_BODY]]:
 ; GFX1200-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1200-NEXT:    [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1200-NEXT:    [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
 ; GFX1200-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 188
 ; GFX1200-NEXT:    [[SCEVGEP1:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
-; GFX1200-NEXT:    [[TMP2:%.*]] = shl i64 [[IV]], 2
+; GFX1200-NEXT:    [[TMP2:%.*]] = shl nuw nsw i64 [[IV]], 2
 ; GFX1200-NEXT:    [[TMP3:%.*]] = add i64 [[TMP2]], 28
 ; GFX1200-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP3]]
 ; GFX1200-NEXT:    [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
@@ -198,10 +198,10 @@ define amdgpu_kernel void @prefetch_two_streams_320B(ptr addrspace(1) nocapture
 ; GFX1250-NEXT:    br label %[[FOR_BODY:.*]]
 ; GFX1250:       [[FOR_BODY]]:
 ; GFX1250-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1250-NEXT:    [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1250-NEXT:    [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
 ; GFX1250-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 348
 ; GFX1250-NEXT:    [[SCEVGEP1:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
-; GFX1250-NEXT:    [[TMP2:%.*]] = shl i64 [[IV]], 2
+; GFX1250-NEXT:    [[TMP2:%.*]] = shl nuw nsw i64 [[IV]], 2
 ; GFX1250-NEXT:    [[TMP3:%.*]] = add i64 [[TMP2]], 28
 ; GFX1250-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP3]]
 ; GFX1250-NEXT:    [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
@@ -226,10 +226,10 @@ define amdgpu_kernel void @prefetch_two_streams_320B(ptr addrspace(1) nocapture
 ; GFX1200-NEXT:    br label %[[FOR_BODY:.*]]
 ; GFX1200:       [[FOR_BODY]]:
 ; GFX1200-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1200-NEXT:    [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1200-NEXT:    [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
 ; GFX1200-NEXT:    [[TMP1:%.*]] = add i64 [[TMP0]], 348
 ; GFX1200-NEXT:    [[SCEVGEP1:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
-; GFX1200-NEXT:    [[TMP2:%.*]] = shl i64 [[IV]], 2
+; GFX1200-NEXT:    [[TMP2:%.*]] = shl nuw nsw i64 [[IV]], 2
 ; GFX1200-NEXT:    [[TMP3:%.*]] = add i64 [[TMP2]], 28
 ; GFX1200-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP3]]
 ; GFX1200-NEXT:    [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
diff --git a/llvm/test/Transforms/LoopIdiom/reuse-lcssa-phi-scev-expansion.ll b/llvm/test/Transforms/LoopIdiom/reuse-lcssa-phi-scev-expansion.ll
index 5011d21d7a044..de6cc5aaa48e4 100644
--- a/llvm/test/Transforms/LoopIdiom/reuse-lcssa-phi-scev-expansion.ll
+++ b/llvm/test/Transforms/LoopIdiom/reuse-lcssa-phi-scev-expansion.ll
@@ -178,14 +178,13 @@ define void @phi_ptr_addressspace_ptrtoint_fail(ptr addrspace(1) %arg) {
 ; CHECK-NEXT:    br i1 false, label %[[LOOP_1]], label %[[LOOP_2_PH:.*]]
 ; CHECK:       [[LOOP_2_PH]]:
 ; CHECK-NEXT:    [[IV_1_LCSSA1:%.*]] = phi i64 [ [[IV_1]], %[[LOOP_1]] ]
-; CHECK-NEXT:    [[IV_1_LCSSA:%.*]] = phi i64 [ [[IV_1]], %[[LOOP_1]] ]
 ; CHECK-NEXT:    [[PHI:%.*]] = phi ptr addrspace(1) [ [[GETELEMENTPTR]], %[[LOOP_1]] ]
 ; CHECK-NEXT:    [[TMP0:%.*]] = shl nuw nsw i64 [[IV_1_LCSSA1]], 2
 ; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[ARG]], i64 [[TMP0]]
 ; CHECK-NEXT:    call void @llvm.memset.p1.i64(ptr addrspace(1) align 4 [[SCEVGEP]], i8 0, i64 8, i1 false)
 ; CHECK-NEXT:    br label %[[LOOP_2_HEADER:.*]]
 ; CHECK:       [[LOOP_2_HEADER]]:
-; CHECK-NEXT:    [[IV_2:%.*]] = phi i64 [ [[IV_1_LCSSA]], %[[LOOP_2_PH]] ], [ [[IV_2_NEXT:%.*]], %[[LOOP_2_LATCH:.*]] ]
+; CHECK-NEXT:    [[IV_2:%.*]] = phi i64 [ [[IV_1_LCSSA1]], %[[LOOP_2_PH]] ], [ [[IV_2_NEXT:%.*]], %[[LOOP_2_LATCH:.*]] ]
 ; CHECK-NEXT:    [[GREP_ARG:%.*]] = getelementptr i32, ptr addrspace(1) [[ARG]], i64 [[IV_2]]
 ; CHECK-NEXT:    [[EC:%.*]] = icmp ult i64 [[IV_2]], 1
 ; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP_2_LATCH]], label %[[EXIT:.*]]
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll b/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
index 344ccf7d03685..49fb58f84a6f0 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
@@ -196,7 +196,7 @@ define void @bound_is_addrec_of_sibling_loop(ptr %a, ptr %b, i64 %n, i64 %d, i1
 ; CHECK-NEXT:  ir-bb<vector.memcheck>:
 ; CHECK-NEXT:    IR   %4 = udiv i64 %n, %d
 ; CHECK-NEXT:    IR   %5 = shl i64 %4, 2
-; CHECK-NEXT:    IR   %6 = shl i64 %iv.1, 2
+; CHECK-NEXT:    IR   %6 = shl i64 %iv.1.lcssa, 2
 ; CHECK-NEXT:    IR   %7 = add i64 %6, %5
 ; CHECK-NEXT:    IR   %scevgep = getelementptr i8, ptr %b, i64 %7
 ; CHECK-NEXT:    IR   %8 = shl i64 %n, 2
diff --git a/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll b/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
index dc7feef799b80..4c51e95d5aa47 100644
--- a/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
+++ b/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
@@ -249,7 +249,7 @@ define void @nested_loop_outer_iv_addrec_invariant_in_inner1(ptr %a, ptr %b, i64
 ; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
 ; CHECK:       [[OUTER_HEADER]]:
 ; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = shl i64 [[OUTER_IV]], 2
+; CHECK-NEXT:    [[TMP1:%.*]] = shl nuw nsw i64 [[OUTER_IV]], 2
 ; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[TMP1]], 4
 ; CHECK-NEXT:    [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP2]]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[OUTER_IV]]
@@ -301,7 +301,7 @@ define void @nested_loop_outer_iv_addrec_invariant_in_inner2(ptr %a, ptr %b, i64
 ; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
 ; CHECK:       [[OUTER_HEADER]]:
 ; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = shl i64 [[OUTER_IV]], 2
+; CHECK-NEXT:    [[TMP1:%.*]] = shl nuw nsw i64 [[OUTER_IV]], 2
 ; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[TMP1]], 4
 ; CHECK-NEXT:    [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP2]]
 ; CHECK-NEXT:    [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[OUTER_IV]]
@@ -356,19 +356,16 @@ define void @nested_loop_bounds_are_scaled_outer_iv(ptr %a, ptr %b, i32 %n, i32
 ; CHECK-NEXT:    [[N_EXT:%.*]] = zext nneg i32 [[N]] to i64
 ; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
 ; CHECK:       [[OUTER_HEADER]]:
-; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ [[INDVAR_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[OUTER_PH]] ]
-; CHECK-NEXT:    [[OUTER_IV1:%.*]] = phi i64 [ 1, %[[OUTER_PH]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ 1, %[[OUTER_PH]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH:%.*]] ]
 ; CHECK-NEXT:    [[TMP0:%.*]] = shl nuw nsw i64 [[OUTER_IV]], 2
-; CHECK-NEXT:    [[TMP4:%.*]] = add i64 [[TMP0]], 4
-; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP0]]
 ; CHECK-NEXT:    [[TMP1:%.*]] = shl nuw nsw i64 [[OUTER_IV]], 3
-; CHECK-NEXT:    [[TMP3:%.*]] = add i64 [[TMP1]], 8
-; CHECK-NEXT:    [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP3]]
-; CHECK-NEXT:    [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP4]]
-; CHECK-NEXT:    [[SCEVGEP4:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP3]]
-; CHECK-NEXT:    [[OUTER_OFF_IS:%.*]] = mul nsw i64 [[OUTER_IV1]], [[IS_EXT]]
-; CHECK-NEXT:    [[OUTER_OFF_JS:%.*]] = mul nsw i64 [[OUTER_IV1]], [[JS_EXT]]
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[OUTER_IV1]], 4
+; CHECK-NEXT:    [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
+; CHECK-NEXT:    [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
+; CHECK-NEXT:    [[SCEVGEP4:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP1]]
+; CHECK-NEXT:    [[OUTER_OFF_IS:%.*]] = mul nsw i64 [[OUTER_IV]], [[IS_EXT]]
+; CHECK-NEXT:    [[OUTER_OFF_JS:%.*]] = mul nsw i64 [[OUTER_IV]], [[JS_EXT]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[OUTER_IV]], 4
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_SCEVCHECK:.*]]
 ; CHECK:       [[VECTOR_SCEVCHECK]]:
 ; CHECK-NEXT:    [[IDENT_CHECK:%.*]] = icmp ne i32 [[JS]], 1
@@ -441,19 +438,18 @@ define void @nested_loop_only_one_bound_is_scaled_outer_iv(ptr %a, ptr %b, i32 %
 ; CHECK-NEXT:    [[N_EXT:%.*]] = zext nneg i32 [[N]] to i64
 ; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
 ; CHECK:       [[OUTER_HEADER]]:
-; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ [[INDVAR_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[OUTER_PH]] ]
-; CHECK-NEXT:    [[OUTER_IV1:%.*]] = phi i64 [ 1, %[[OUTER_PH]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
+; CHECK-NEXT:    [[INDVAR:%.*]] = phi i64 [ [[INDVAR_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[OUTER_PH]] ]
+; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ 1, %[[OUTER_PH]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
 ; CHECK-NEXT:    [[TMP0:%.*]] = mul nuw nsw i64 [[OUTER_IV]], 12
-; CHECK-NEXT:    [[TMP4:%.*]] = add i64 [[TMP0]], 12
-; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP4]]
-; CHECK-NEXT:    [[TMP1:%.*]] = mul nuw nsw i64 [[OUTER_IV]], 24
+; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP1:%.*]] = mul nuw nsw i64 [[INDVAR]], 24
 ; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[TMP1]], 16
 ; CHECK-NEXT:    [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP2]]
-; CHECK-NEXT:    [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP4]]
+; CHECK-NEXT:    [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
 ; CHECK-NEXT:    [[SCEVGEP4:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP2]]
-; CHECK-NEXT:    [[OUTER_OFF_IS:%.*]] = mul nsw i64 [[OUTER_IV1]], [[IS_EXT]]
-; CHECK-NEXT:    [[OUTER_OFF_JS:%.*]] = mul nsw i64 [[OUTER_IV1]], [[JS_EXT]]
-; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[OUTER_IV1]], 4
+; CHECK-NEXT:    [[OUTER_OFF_IS:%.*]] = mul nsw i64 [[OUTER_IV]], [[IS_EXT]]
+; CHECK-NEXT:    [[OUTER_OFF_JS:%.*]] = mul nsw i64 [[OUTER_IV]], [[JS_EXT]]
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[OUTER_IV]], 4
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_SCEVCHECK:.*]]
 ; CHECK:       [[VECTOR_SCEVCHECK]]:
 ; CHECK-NEXT:    [[IDENT_CHECK:%.*]] = icmp ne i32 [[JS]], 1
@@ -529,7 +525,7 @@ define void @nested_loop_bound_needs_constant_correction(ptr %a, ptr %b, i64 %n)
 ; CHECK-NEXT:    [[TMP4:%.*]] = shl i64 [[OUTER_IV]], 2
 ; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[TMP5:%.*]] = shl i64 [[OUTER_IV]], 3
-; CHECK-NEXT:    [[TMP6:%.*]] = add i64 [[TMP5]], 4
+; CHECK-NEXT:    [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 4
 ; CHECK-NEXT:    [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP6]]
 ; CHECK-NEXT:    [[TMP3:%.*]] = sub i64 [[N]], [[OUTER_IV]]
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP3]], 4
@@ -583,7 +579,7 @@ define void @nested_loop_bound_reuses_scaled_iv(ptr %a, ptr %b, ptr %c, i64 %n)
 ; CHECK-NEXT:    [[SCALED_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[SCALED_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
 ; CHECK-NEXT:    [[TMP2:%.*]] = shl i64 [[OUTER_IV]], 2
 ; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP2]]
-; CHECK-NEXT:    [[TMP3:%.*]] = add i64 [[SCALED_IV]], 4
+; CHECK-NEXT:    [[TMP3:%.*]] = add nuw nsw i64 [[SCALED_IV]], 4
 ; CHECK-NEXT:    [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP3]]
 ; CHECK-NEXT:    [[TMP4:%.*]] = sub i64 [[N]], [[OUTER_IV]]
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP4]], 4
@@ -633,17 +629,15 @@ define void @nested_loop_bound_with_negative_step(ptr %a, ptr %b, i64 %n) {
 ; CHECK-NEXT:    [[TMP0:%.*]] = shl i64 [[N]], 2
 ; CHECK-NEXT:    [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
 ; CHECK-NEXT:    [[TMP1:%.*]] = shl i64 [[N]], 3
-; CHECK-NEXT:    [[TMP6:%.*]] = add nuw nsw i64 [[TMP1]], 4
 ; CHECK-NEXT:    [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
 ; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
 ; CHECK:       [[OUTER_HEADER]]:
 ; CHECK-NEXT:    [[INDVAR:%.*]] = phi i64 [ [[INDVAR_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[ENTRY]] ]
 ; CHECK-NEXT:    [[OUTER_IV:%.*]] = phi i64 [ [[N]], %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
-; CHECK-NEXT:    [[TMP3:%.*]] = mul i64 [[INDVAR]], -4
-; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[TMP0]], [[TMP3]]
+; CHECK-NEXT:    [[TMP2:%.*]] = shl i64 [[OUTER_IV]], 2
 ; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP2]]
-; CHECK-NEXT:    [[TMP5:%.*]] = mul i64 [[INDVAR]], -8
-; CHECK-NEXT:    [[TMP4:%.*]] = add i64 [[TMP6]], [[TMP5]]
+; CHECK-NEXT:    [[TMP3:%.*]] = shl i64 [[OUTER_IV]], 3
+; CHECK-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[TMP3]], 4
 ; CHECK-NEXT:    [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP4]]
 ; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[INDVAR]], 4
 ; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_MEMCHECK:.*]]



More information about the llvm-commits mailing list