[llvm] [SCEVExpander] Reuse unscaled recurrence when expanding scaled one. (PR #223850)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 22 02:30:19 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/223850
>From 12187b99acf258a61d5dafc4022c1071da7d19ac Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 15 Sep 2026 22:27:12 +0100
Subject: [PATCH 1/2] Precommit test
---
.../runtime-checks-difference.ll | 113 ++++++++++++++++++
1 file changed, 113 insertions(+)
diff --git a/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll b/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
index e816b559bffe6..dc7feef799b80 100644
--- a/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
+++ b/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
@@ -566,6 +566,119 @@ exit:
ret void
}
+; Same as @nested_loop_bound_needs_constant_correction, but the loop already
+; has an induction variable stepping by 8. The bound is expanded as an offset
+; from it.
+define void @nested_loop_bound_reuses_scaled_iv(ptr %a, ptr %b, ptr %c, i64 %n) {
+; CHECK-LABEL: define void @nested_loop_bound_reuses_scaled_iv(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], ptr [[C:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[N]], 2
+; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
+; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[N]], 3
+; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
+; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
+; CHECK: [[OUTER_HEADER]]:
+; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH:%.*]] ]
+; CHECK-NEXT: [[SCALED_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[SCALED_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = shl i64 [[OUTER_IV]], 2
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP2]]
+; CHECK-NEXT: [[TMP3:%.*]] = add i64 [[SCALED_IV]], 4
+; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP3]]
+; CHECK-NEXT: [[TMP4:%.*]] = sub i64 [[N]], [[OUTER_IV]]
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP4]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_MEMCHECK:.*]]
+; CHECK: [[VECTOR_MEMCHECK]]:
+; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP3]]
+; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[SCEVGEP2]], [[SCEVGEP1]]
+; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
+; CHECK-NEXT: br i1 [[FOUND_CONFLICT]], [[SCALAR_PH]], [[VECTOR_PH:label %.*]]
+;
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+ %scaled.iv = phi i64 [ 0, %entry ], [ %scaled.iv.next, %outer.latch ]
+ br label %inner.body
+
+inner.body:
+ %inner.iv = phi i64 [ %outer.iv, %outer.header ], [ %inner.iv.next, %inner.body ]
+ %gep.a = getelementptr inbounds { i32, i32 }, ptr %a, i64 %inner.iv, i32 1
+ %l = load i32, ptr %gep.a, align 4
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %inner.iv
+ store i32 %l, ptr %gep.b, align 4
+ %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+ %inner.cond = icmp eq i64 %inner.iv.next, %n
+ br i1 %inner.cond, label %outer.latch, label %inner.body
+
+outer.latch:
+ %gep.c = getelementptr inbounds i8, ptr %c, i64 %scaled.iv
+ store i64 0, ptr %gep.c, align 8
+ %scaled.iv.next = add nuw nsw i64 %scaled.iv, 8
+ %outer.iv.next = add nuw nsw i64 %outer.iv, 1
+ %outer.cond = icmp eq i64 %outer.iv.next, %n
+ br i1 %outer.cond, label %exit, label %outer.header
+
+exit:
+ ret void
+}
+
+; Same as @nested_loop_bound_needs_constant_correction, but with an outer
+; induction variable counting down. The bounds are expanded by scaling the IV.
+define void @nested_loop_bound_with_negative_step(ptr %a, ptr %b, i64 %n) {
+; CHECK-LABEL: define void @nested_loop_bound_with_negative_step(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[N]], 2
+; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
+; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[N]], 3
+; CHECK-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP1]], 4
+; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
+; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
+; CHECK: [[OUTER_HEADER]]:
+; CHECK-NEXT: [[INDVAR:%.*]] = phi i64 [ [[INDVAR_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ [[N]], %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = mul i64 [[INDVAR]], -4
+; CHECK-NEXT: [[TMP2:%.*]] = add i64 [[TMP0]], [[TMP3]]
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP2]]
+; CHECK-NEXT: [[TMP5:%.*]] = mul i64 [[INDVAR]], -8
+; CHECK-NEXT: [[TMP4:%.*]] = add i64 [[TMP6]], [[TMP5]]
+; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[INDVAR]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_MEMCHECK:.*]]
+; CHECK: [[VECTOR_MEMCHECK]]:
+; CHECK-NEXT: [[BOUND0:%.*]] = icmp ult ptr [[SCEVGEP]], [[SCEVGEP3]]
+; CHECK-NEXT: [[BOUND1:%.*]] = icmp ult ptr [[SCEVGEP2]], [[SCEVGEP1]]
+; CHECK-NEXT: [[FOUND_CONFLICT:%.*]] = and i1 [[BOUND0]], [[BOUND1]]
+; CHECK-NEXT: br i1 [[FOUND_CONFLICT]], [[SCALAR_PH]], [[VECTOR_PH:label %.*]]
+;
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ %n, %entry ], [ %outer.iv.next, %outer.latch ]
+ br label %inner.body
+
+inner.body:
+ %inner.iv = phi i64 [ %outer.iv, %outer.header ], [ %inner.iv.next, %inner.body ]
+ %gep.a = getelementptr inbounds { i32, i32 }, ptr %a, i64 %inner.iv, i32 1
+ %l = load i32, ptr %gep.a, align 4
+ %gep.b = getelementptr inbounds i32, ptr %b, i64 %inner.iv
+ store i32 %l, ptr %gep.b, align 4
+ %inner.iv.next = add nuw nsw i64 %inner.iv, 1
+ %inner.cond = icmp eq i64 %inner.iv.next, %n
+ br i1 %inner.cond, label %outer.latch, label %inner.body
+
+outer.latch:
+ %outer.iv.next = add nsw i64 %outer.iv, -1
+ %outer.cond = icmp eq i64 %outer.iv.next, 0
+ br i1 %outer.cond, label %exit, label %outer.header
+
+exit:
+ ret void
+}
+
; Test case where the AddRec for the pointers in the inner loop have the AddRec
; of the outer loop as start value. It is sufficient to subtract the start
; values (%dst, %src) of the outer AddRecs.
>From 5ba772b37715b93e1713cf70f6cd1d589dde1b06 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 19 Aug 2026 13:16:46 +0100
Subject: [PATCH 2/2] [SCEVExpander] Reuse unscaled recurrence when expanding
scaled one.
When expanding a scaled recurrence, like {8*X,+,8}, we can re-use an
existing {X,+,1} recurrence and just scale the existing result. This
avoids introducing new IVs in loops unnecessarily.
This improves SCEV expansions in a number of cases:
https://github.com/dtcxzyw/llvm-opt-benchmark-nightly/pull/1320.
Found while investigating small SCEV expansion regressions due to
additional folding/flag inference in ConstraintElimination.
---
.../Utils/ScalarEvolutionExpander.h | 1 +
.../Utils/ScalarEvolutionExpander.cpp | 55 +++++++++++++++++++
.../AMDGPU/vmem-cache-line-size.ll | 20 +++----
.../reuse-lcssa-phi-scev-expansion.ll | 3 +-
.../LoopVectorize/VPlan/memory-checks.ll | 2 +-
.../runtime-checks-difference.ll | 52 ++++++++----------
6 files changed, 91 insertions(+), 42 deletions(-)
diff --git a/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h b/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h
index c98c0cb52fa9c..6886342721399 100644
--- a/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h
+++ b/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h
@@ -561,6 +561,7 @@ class SCEVExpander : public SCEVUseVisitor<SCEVExpander, Value *> {
bool isExpandedAddRecExprPHI(PHINode *PN, Instruction *IncV, const Loop *L);
Value *tryToReuseLCSSAPhi(SCEVUseT<const SCEVAddRecExpr *> S);
+ Value *tryToReuseScaledAddRec(const SCEVAddRecExpr *S);
Value *expandAddRecExprLiterally(SCEVUseT<const SCEVAddRecExpr *> S);
PHINode *getAddRecExprPHILiterally(const SCEVAddRecExpr *Normalized,
const Loop *L, Type *&TruncTy,
diff --git a/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp b/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
index 93baf01173f98..3d3de1ff7020c 100644
--- a/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
+++ b/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
@@ -17,6 +17,7 @@
#include "llvm/ADT/ScopeExit.h"
#include "llvm/Analysis/InstructionSimplify.h"
#include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/ScalarEvolutionDivision.h"
#include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/Analysis/ValueTracking.h"
@@ -1334,6 +1335,57 @@ Value *SCEVExpander::tryToReuseLCSSAPhi(SCEVUseT<const SCEVAddRecExpr *> S) {
return nullptr;
}
+/// Try to re-use an existing AddRec when expanding S, if
+/// AddRec * Scale + Offset == S for constant Scale and Offset.
+Value *SCEVExpander::tryToReuseScaledAddRec(const SCEVAddRecExpr *S) {
+ const APInt *Step;
+ if (!S->getType()->isIntegerTy() ||
+ !match(S->getStepRecurrence(SE), m_scev_APInt(Step)))
+ return nullptr;
+
+ APInt Factor = Step->abs();
+ if (Factor.ule(1))
+ return nullptr;
+
+ const SCEV *FactorS = SE.getConstant(Factor);
+
+ // Split S into Divided * FactorS + Offset, with a constant Offset, and check
+ // if there is an existing Divided that can be scaled back to S.
+ const SCEV *Divided, *Offset;
+ SCEVDivision::divide(SE, S, FactorS, &Divided, &Offset);
+ if (!isa<SCEVConstant>(Offset))
+ return nullptr;
+ const SCEV *Scaled = SE.getMulExpr(Divided, FactorS);
+ if (SE.getAddExpr(Scaled, Offset) != S)
+ return nullptr;
+
+ // First check if we have an existing expansion for Scaled directly. Otherwise
+ // look for Divided and scale manually.
+ const Instruction *InsertPt = &*Builder.GetInsertPoint();
+ Value *V = nullptr;
+ if (!Offset->isZero())
+ V = findExistingExpansionAndDropPoisonFlags(Scaled, InsertPt);
+
+ if (!V) {
+ Value *Base = findExistingExpansionAndDropPoisonFlags(Divided, InsertPt);
+ if (!Base)
+ return nullptr;
+ // Carry over the no-wrap facts that hold for the scaling, otherwise the new
+ // expansion may miss flags the original one had.
+ SCEV::NoWrapFlags Flags = SCEV::FlagAnyWrap;
+ if (SE.willNotOverflow(Instruction::Mul, /*Signed=*/false, Divided,
+ FactorS))
+ Flags = ScalarEvolution::setFlags(Flags, SCEV::FlagNUW);
+ if (SE.willNotOverflow(Instruction::Mul, /*Signed=*/true, Divided, FactorS))
+ Flags = ScalarEvolution::setFlags(Flags, SCEV::FlagNSW);
+ V = expand(SCEVUse(SE.getMulExpr(SE.getUnknown(Base), FactorS), Flags));
+ }
+
+ if (Offset->isZero())
+ return V;
+ return expand(SE.getAddExpr(SE.getUnknown(V), Offset));
+}
+
Value *SCEVExpander::visitAddRecExpr(SCEVUseT<const SCEVAddRecExpr *> S) {
// In canonical mode we compute the addrec as an expression of a canonical IV
// using evaluateAtIteration and expand the resulting SCEV expression. This
@@ -1348,6 +1400,9 @@ Value *SCEVExpander::visitAddRecExpr(SCEVUseT<const SCEVAddRecExpr *> S) {
if (!CanonicalMode || (S->getNumOperands() > 2))
return expandAddRecExprLiterally(S);
+ if (Value *V = tryToReuseScaledAddRec(S))
+ return V;
+
Type *Ty = SE.getEffectiveSCEVType(S->getType());
const Loop *L = S->getLoop();
diff --git a/llvm/test/Transforms/LoopDataPrefetch/AMDGPU/vmem-cache-line-size.ll b/llvm/test/Transforms/LoopDataPrefetch/AMDGPU/vmem-cache-line-size.ll
index 8cca57c170eee..ba65b202387ec 100644
--- a/llvm/test/Transforms/LoopDataPrefetch/AMDGPU/vmem-cache-line-size.ll
+++ b/llvm/test/Transforms/LoopDataPrefetch/AMDGPU/vmem-cache-line-size.ll
@@ -10,7 +10,7 @@ define amdgpu_kernel void @prefetch_two_streams_80B(ptr addrspace(1) nocapture %
; GFX1250-NEXT: br label %[[FOR_BODY:.*]]
; GFX1250: [[FOR_BODY]]:
; GFX1250-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1250-NEXT: [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1250-NEXT: [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
; GFX1250-NEXT: [[TMP1:%.*]] = add i64 [[TMP0]], 28
; GFX1250-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
; GFX1250-NEXT: [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
@@ -34,7 +34,7 @@ define amdgpu_kernel void @prefetch_two_streams_80B(ptr addrspace(1) nocapture %
; GFX1200-NEXT: br label %[[FOR_BODY:.*]]
; GFX1200: [[FOR_BODY]]:
; GFX1200-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1200-NEXT: [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1200-NEXT: [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
; GFX1200-NEXT: [[TMP1:%.*]] = add i64 [[TMP0]], 28
; GFX1200-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
; GFX1200-NEXT: [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
@@ -100,10 +100,10 @@ define amdgpu_kernel void @prefetch_two_streams_160B(ptr addrspace(1) nocapture
; GFX1250-NEXT: br label %[[FOR_BODY:.*]]
; GFX1250: [[FOR_BODY]]:
; GFX1250-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1250-NEXT: [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1250-NEXT: [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
; GFX1250-NEXT: [[TMP1:%.*]] = add i64 [[TMP0]], 188
; GFX1250-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
-; GFX1250-NEXT: [[TMP2:%.*]] = shl i64 [[IV]], 2
+; GFX1250-NEXT: [[TMP2:%.*]] = shl nuw nsw i64 [[IV]], 2
; GFX1250-NEXT: [[TMP3:%.*]] = add i64 [[TMP2]], 28
; GFX1250-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP3]]
; GFX1250-NEXT: [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
@@ -128,10 +128,10 @@ define amdgpu_kernel void @prefetch_two_streams_160B(ptr addrspace(1) nocapture
; GFX1200-NEXT: br label %[[FOR_BODY:.*]]
; GFX1200: [[FOR_BODY]]:
; GFX1200-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1200-NEXT: [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1200-NEXT: [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
; GFX1200-NEXT: [[TMP1:%.*]] = add i64 [[TMP0]], 188
; GFX1200-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
-; GFX1200-NEXT: [[TMP2:%.*]] = shl i64 [[IV]], 2
+; GFX1200-NEXT: [[TMP2:%.*]] = shl nuw nsw i64 [[IV]], 2
; GFX1200-NEXT: [[TMP3:%.*]] = add i64 [[TMP2]], 28
; GFX1200-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP3]]
; GFX1200-NEXT: [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
@@ -198,10 +198,10 @@ define amdgpu_kernel void @prefetch_two_streams_320B(ptr addrspace(1) nocapture
; GFX1250-NEXT: br label %[[FOR_BODY:.*]]
; GFX1250: [[FOR_BODY]]:
; GFX1250-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1250-NEXT: [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1250-NEXT: [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
; GFX1250-NEXT: [[TMP1:%.*]] = add i64 [[TMP0]], 348
; GFX1250-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
-; GFX1250-NEXT: [[TMP2:%.*]] = shl i64 [[IV]], 2
+; GFX1250-NEXT: [[TMP2:%.*]] = shl nuw nsw i64 [[IV]], 2
; GFX1250-NEXT: [[TMP3:%.*]] = add i64 [[TMP2]], 28
; GFX1250-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP3]]
; GFX1250-NEXT: [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
@@ -226,10 +226,10 @@ define amdgpu_kernel void @prefetch_two_streams_320B(ptr addrspace(1) nocapture
; GFX1200-NEXT: br label %[[FOR_BODY:.*]]
; GFX1200: [[FOR_BODY]]:
; GFX1200-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
-; GFX1200-NEXT: [[TMP0:%.*]] = shl i64 [[IV]], 2
+; GFX1200-NEXT: [[TMP0:%.*]] = shl nuw nsw i64 [[IV]], 2
; GFX1200-NEXT: [[TMP1:%.*]] = add i64 [[TMP0]], 348
; GFX1200-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP1]]
-; GFX1200-NEXT: [[TMP2:%.*]] = shl i64 [[IV]], 2
+; GFX1200-NEXT: [[TMP2:%.*]] = shl nuw nsw i64 [[IV]], 2
; GFX1200-NEXT: [[TMP3:%.*]] = add i64 [[TMP2]], 28
; GFX1200-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[B]], i64 [[TMP3]]
; GFX1200-NEXT: [[IDX0:%.*]] = getelementptr inbounds i32, ptr addrspace(1) [[B]], i64 [[IV]]
diff --git a/llvm/test/Transforms/LoopIdiom/reuse-lcssa-phi-scev-expansion.ll b/llvm/test/Transforms/LoopIdiom/reuse-lcssa-phi-scev-expansion.ll
index 5011d21d7a044..de6cc5aaa48e4 100644
--- a/llvm/test/Transforms/LoopIdiom/reuse-lcssa-phi-scev-expansion.ll
+++ b/llvm/test/Transforms/LoopIdiom/reuse-lcssa-phi-scev-expansion.ll
@@ -178,14 +178,13 @@ define void @phi_ptr_addressspace_ptrtoint_fail(ptr addrspace(1) %arg) {
; CHECK-NEXT: br i1 false, label %[[LOOP_1]], label %[[LOOP_2_PH:.*]]
; CHECK: [[LOOP_2_PH]]:
; CHECK-NEXT: [[IV_1_LCSSA1:%.*]] = phi i64 [ [[IV_1]], %[[LOOP_1]] ]
-; CHECK-NEXT: [[IV_1_LCSSA:%.*]] = phi i64 [ [[IV_1]], %[[LOOP_1]] ]
; CHECK-NEXT: [[PHI:%.*]] = phi ptr addrspace(1) [ [[GETELEMENTPTR]], %[[LOOP_1]] ]
; CHECK-NEXT: [[TMP0:%.*]] = shl nuw nsw i64 [[IV_1_LCSSA1]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr addrspace(1) [[ARG]], i64 [[TMP0]]
; CHECK-NEXT: call void @llvm.memset.p1.i64(ptr addrspace(1) align 4 [[SCEVGEP]], i8 0, i64 8, i1 false)
; CHECK-NEXT: br label %[[LOOP_2_HEADER:.*]]
; CHECK: [[LOOP_2_HEADER]]:
-; CHECK-NEXT: [[IV_2:%.*]] = phi i64 [ [[IV_1_LCSSA]], %[[LOOP_2_PH]] ], [ [[IV_2_NEXT:%.*]], %[[LOOP_2_LATCH:.*]] ]
+; CHECK-NEXT: [[IV_2:%.*]] = phi i64 [ [[IV_1_LCSSA1]], %[[LOOP_2_PH]] ], [ [[IV_2_NEXT:%.*]], %[[LOOP_2_LATCH:.*]] ]
; CHECK-NEXT: [[GREP_ARG:%.*]] = getelementptr i32, ptr addrspace(1) [[ARG]], i64 [[IV_2]]
; CHECK-NEXT: [[EC:%.*]] = icmp ult i64 [[IV_2]], 1
; CHECK-NEXT: br i1 [[EC]], label %[[LOOP_2_LATCH]], label %[[EXIT:.*]]
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll b/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
index 344ccf7d03685..49fb58f84a6f0 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/memory-checks.ll
@@ -196,7 +196,7 @@ define void @bound_is_addrec_of_sibling_loop(ptr %a, ptr %b, i64 %n, i64 %d, i1
; CHECK-NEXT: ir-bb<vector.memcheck>:
; CHECK-NEXT: IR %4 = udiv i64 %n, %d
; CHECK-NEXT: IR %5 = shl i64 %4, 2
-; CHECK-NEXT: IR %6 = shl i64 %iv.1, 2
+; CHECK-NEXT: IR %6 = shl i64 %iv.1.lcssa, 2
; CHECK-NEXT: IR %7 = add i64 %6, %5
; CHECK-NEXT: IR %scevgep = getelementptr i8, ptr %b, i64 %7
; CHECK-NEXT: IR %8 = shl i64 %n, 2
diff --git a/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll b/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
index dc7feef799b80..4c51e95d5aa47 100644
--- a/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
+++ b/llvm/test/Transforms/LoopVectorize/runtime-checks-difference.ll
@@ -249,7 +249,7 @@ define void @nested_loop_outer_iv_addrec_invariant_in_inner1(ptr %a, ptr %b, i64
; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
; CHECK: [[OUTER_HEADER]]:
; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[OUTER_IV]], 2
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw nsw i64 [[OUTER_IV]], 2
; CHECK-NEXT: [[TMP2:%.*]] = add i64 [[TMP1]], 4
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP2]]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[OUTER_IV]]
@@ -301,7 +301,7 @@ define void @nested_loop_outer_iv_addrec_invariant_in_inner2(ptr %a, ptr %b, i64
; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
; CHECK: [[OUTER_HEADER]]:
; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[OUTER_IV]], 2
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw nsw i64 [[OUTER_IV]], 2
; CHECK-NEXT: [[TMP2:%.*]] = add i64 [[TMP1]], 4
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP2]]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i32, ptr [[A]], i64 [[OUTER_IV]]
@@ -356,19 +356,16 @@ define void @nested_loop_bounds_are_scaled_outer_iv(ptr %a, ptr %b, i32 %n, i32
; CHECK-NEXT: [[N_EXT:%.*]] = zext nneg i32 [[N]] to i64
; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
; CHECK: [[OUTER_HEADER]]:
-; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ [[INDVAR_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[OUTER_PH]] ]
-; CHECK-NEXT: [[OUTER_IV1:%.*]] = phi i64 [ 1, %[[OUTER_PH]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
+; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ 1, %[[OUTER_PH]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH:%.*]] ]
; CHECK-NEXT: [[TMP0:%.*]] = shl nuw nsw i64 [[OUTER_IV]], 2
-; CHECK-NEXT: [[TMP4:%.*]] = add i64 [[TMP0]], 4
-; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP4]]
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP0]]
; CHECK-NEXT: [[TMP1:%.*]] = shl nuw nsw i64 [[OUTER_IV]], 3
-; CHECK-NEXT: [[TMP3:%.*]] = add i64 [[TMP1]], 8
-; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP3]]
-; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP4]]
-; CHECK-NEXT: [[SCEVGEP4:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP3]]
-; CHECK-NEXT: [[OUTER_OFF_IS:%.*]] = mul nsw i64 [[OUTER_IV1]], [[IS_EXT]]
-; CHECK-NEXT: [[OUTER_OFF_JS:%.*]] = mul nsw i64 [[OUTER_IV1]], [[JS_EXT]]
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[OUTER_IV1]], 4
+; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
+; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
+; CHECK-NEXT: [[SCEVGEP4:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP1]]
+; CHECK-NEXT: [[OUTER_OFF_IS:%.*]] = mul nsw i64 [[OUTER_IV]], [[IS_EXT]]
+; CHECK-NEXT: [[OUTER_OFF_JS:%.*]] = mul nsw i64 [[OUTER_IV]], [[JS_EXT]]
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[OUTER_IV]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_SCEVCHECK:.*]]
; CHECK: [[VECTOR_SCEVCHECK]]:
; CHECK-NEXT: [[IDENT_CHECK:%.*]] = icmp ne i32 [[JS]], 1
@@ -441,19 +438,18 @@ define void @nested_loop_only_one_bound_is_scaled_outer_iv(ptr %a, ptr %b, i32 %
; CHECK-NEXT: [[N_EXT:%.*]] = zext nneg i32 [[N]] to i64
; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
; CHECK: [[OUTER_HEADER]]:
-; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ [[INDVAR_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[OUTER_PH]] ]
-; CHECK-NEXT: [[OUTER_IV1:%.*]] = phi i64 [ 1, %[[OUTER_PH]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
+; CHECK-NEXT: [[INDVAR:%.*]] = phi i64 [ [[INDVAR_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[OUTER_PH]] ]
+; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ 1, %[[OUTER_PH]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
; CHECK-NEXT: [[TMP0:%.*]] = mul nuw nsw i64 [[OUTER_IV]], 12
-; CHECK-NEXT: [[TMP4:%.*]] = add i64 [[TMP0]], 12
-; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP4]]
-; CHECK-NEXT: [[TMP1:%.*]] = mul nuw nsw i64 [[OUTER_IV]], 24
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP0]]
+; CHECK-NEXT: [[TMP1:%.*]] = mul nuw nsw i64 [[INDVAR]], 24
; CHECK-NEXT: [[TMP2:%.*]] = add i64 [[TMP1]], 16
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP2]]
-; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP4]]
+; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
; CHECK-NEXT: [[SCEVGEP4:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP2]]
-; CHECK-NEXT: [[OUTER_OFF_IS:%.*]] = mul nsw i64 [[OUTER_IV1]], [[IS_EXT]]
-; CHECK-NEXT: [[OUTER_OFF_JS:%.*]] = mul nsw i64 [[OUTER_IV1]], [[JS_EXT]]
-; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[OUTER_IV1]], 4
+; CHECK-NEXT: [[OUTER_OFF_IS:%.*]] = mul nsw i64 [[OUTER_IV]], [[IS_EXT]]
+; CHECK-NEXT: [[OUTER_OFF_JS:%.*]] = mul nsw i64 [[OUTER_IV]], [[JS_EXT]]
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[OUTER_IV]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_SCEVCHECK:.*]]
; CHECK: [[VECTOR_SCEVCHECK]]:
; CHECK-NEXT: [[IDENT_CHECK:%.*]] = icmp ne i32 [[JS]], 1
@@ -529,7 +525,7 @@ define void @nested_loop_bound_needs_constant_correction(ptr %a, ptr %b, i64 %n)
; CHECK-NEXT: [[TMP4:%.*]] = shl i64 [[OUTER_IV]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP4]]
; CHECK-NEXT: [[TMP5:%.*]] = shl i64 [[OUTER_IV]], 3
-; CHECK-NEXT: [[TMP6:%.*]] = add i64 [[TMP5]], 4
+; CHECK-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP5]], 4
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP6]]
; CHECK-NEXT: [[TMP3:%.*]] = sub i64 [[N]], [[OUTER_IV]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP3]], 4
@@ -583,7 +579,7 @@ define void @nested_loop_bound_reuses_scaled_iv(ptr %a, ptr %b, ptr %c, i64 %n)
; CHECK-NEXT: [[SCALED_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[SCALED_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
; CHECK-NEXT: [[TMP2:%.*]] = shl i64 [[OUTER_IV]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP2]]
-; CHECK-NEXT: [[TMP3:%.*]] = add i64 [[SCALED_IV]], 4
+; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[SCALED_IV]], 4
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP3]]
; CHECK-NEXT: [[TMP4:%.*]] = sub i64 [[N]], [[OUTER_IV]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP4]], 4
@@ -633,17 +629,15 @@ define void @nested_loop_bound_with_negative_step(ptr %a, ptr %b, i64 %n) {
; CHECK-NEXT: [[TMP0:%.*]] = shl i64 [[N]], 2
; CHECK-NEXT: [[SCEVGEP1:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP0]]
; CHECK-NEXT: [[TMP1:%.*]] = shl i64 [[N]], 3
-; CHECK-NEXT: [[TMP6:%.*]] = add nuw nsw i64 [[TMP1]], 4
; CHECK-NEXT: [[SCEVGEP3:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP1]]
; CHECK-NEXT: br label %[[OUTER_HEADER:.*]]
; CHECK: [[OUTER_HEADER]]:
; CHECK-NEXT: [[INDVAR:%.*]] = phi i64 [ [[INDVAR_NEXT:%.*]], [[OUTER_LATCH:%.*]] ], [ 0, %[[ENTRY]] ]
; CHECK-NEXT: [[OUTER_IV:%.*]] = phi i64 [ [[N]], %[[ENTRY]] ], [ [[OUTER_IV_NEXT:%.*]], [[OUTER_LATCH]] ]
-; CHECK-NEXT: [[TMP3:%.*]] = mul i64 [[INDVAR]], -4
-; CHECK-NEXT: [[TMP2:%.*]] = add i64 [[TMP0]], [[TMP3]]
+; CHECK-NEXT: [[TMP2:%.*]] = shl i64 [[OUTER_IV]], 2
; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[B]], i64 [[TMP2]]
-; CHECK-NEXT: [[TMP5:%.*]] = mul i64 [[INDVAR]], -8
-; CHECK-NEXT: [[TMP4:%.*]] = add i64 [[TMP6]], [[TMP5]]
+; CHECK-NEXT: [[TMP3:%.*]] = shl i64 [[OUTER_IV]], 3
+; CHECK-NEXT: [[TMP4:%.*]] = add nuw nsw i64 [[TMP3]], 4
; CHECK-NEXT: [[SCEVGEP2:%.*]] = getelementptr i8, ptr [[A]], i64 [[TMP4]]
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[INDVAR]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], [[SCALAR_PH:label %.*]], label %[[VECTOR_MEMCHECK:.*]]
More information about the llvm-commits
mailing list