[llvm] [SCEV] Use howManyLessThans to implement howManyGreaterThans. (PR #226846)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 03:12:23 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/226846
>From 1bb6c46d5b956255233bacc275729fa3a3af0efb Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Tue, 22 Sep 2026 18:33:26 +0100
Subject: [PATCH 1/2] Precommit tests
---
.../exit-count-greater-than.ll | 235 ++++++++++++++++++
1 file changed, 235 insertions(+)
diff --git a/llvm/test/Analysis/ScalarEvolution/exit-count-greater-than.ll b/llvm/test/Analysis/ScalarEvolution/exit-count-greater-than.ll
index e976506d87619..5cdad4418d45b 100644
--- a/llvm/test/Analysis/ScalarEvolution/exit-count-greater-than.ll
+++ b/llvm/test/Analysis/ScalarEvolution/exit-count-greater-than.ll
@@ -234,3 +234,238 @@ loop:
exit:
ret void
}
+
+define void @sgt_guarded_sext_start_and_bound(i32 %start, i32 %bound) {
+; CHECK-LABEL: 'sgt_guarded_sext_start_and_bound'
+; CHECK-NEXT: Determining loop execution counts for: @sgt_guarded_sext_start_and_bound
+; CHECK-NEXT: Loop %loop: backedge-taken count is ((sext i32 %start to i64) + (-1 * (sext i32 %bound to i64))<nsw>)
+; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i64 4294967295
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is ((sext i32 %start to i64) + (-1 * (sext i32 %bound to i64))<nsw>)
+; CHECK-NEXT: Loop %loop: Trip multiple is 1
+;
+entry:
+ %guard = icmp slt i32 %start, %bound
+ br i1 %guard, label %exit, label %ph
+
+ph:
+ %start.ext = sext i32 %start to i64
+ %bound.ext = sext i32 %bound to i64
+ br label %loop
+
+loop:
+ %iv = phi i64 [ %start.ext, %ph ], [ %iv.next, %loop ]
+ %iv.next = add nsw i64 %iv, -1
+ %ec = icmp sgt i64 %iv, %bound.ext
+ br i1 %ec, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; As above, but the start is zero-extended and the bound sign-extended.
+define void @sgt_guarded_zext_start(i32 %start, i32 %bound) {
+; CHECK-LABEL: 'sgt_guarded_zext_start'
+; CHECK-NEXT: Determining loop execution counts for: @sgt_guarded_zext_start
+; CHECK-NEXT: Loop %loop: backedge-taken count is ((zext i32 %start to i64) + (-1 * (sext i32 %bound to i64))<nsw>)
+; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i64 6442450943
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is ((zext i32 %start to i64) + (-1 * (sext i32 %bound to i64))<nsw>)
+; CHECK-NEXT: Loop %loop: Trip multiple is 1
+;
+entry:
+ %guard = icmp slt i32 %start, %bound
+ br i1 %guard, label %exit, label %ph
+
+ph:
+ %start.ext = zext i32 %start to i64
+ %bound.ext = sext i32 %bound to i64
+ br label %loop
+
+loop:
+ %iv = phi i64 [ %start.ext, %ph ], [ %iv.next, %loop ]
+ %iv.next = add nsw i64 %iv, -1
+ %ec = icmp sgt i64 %iv, %bound.ext
+ br i1 %ec, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+define void @ugt_guarded_bound_is_add(i32 %x, i32 %a, i32 %b) {
+; CHECK-LABEL: 'ugt_guarded_bound_is_add'
+; CHECK-NEXT: Determining loop execution counts for: @ugt_guarded_bound_is_add
+; CHECK-NEXT: Loop %loop: backedge-taken count is (-1 + (zext i8 (trunc i32 %x to i8) to i32) + (-1 * (%a + %b)))
+; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i32 -1
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is (-1 + (zext i8 (trunc i32 %x to i8) to i32) + (-1 * (%a + %b)))
+; CHECK-NEXT: Loop %loop: Trip multiple is 1
+;
+entry:
+ %n = and i32 %x, 255
+ %lim = add i32 %a, %b
+ %start = add nsw i32 %n, -1
+ %guard = icmp ugt i32 %n, %lim
+ br i1 %guard, label %loop, label %exit
+
+loop:
+ %iv = phi i32 [ %start, %entry ], [ %iv.next, %loop ]
+ %iv.next = add i32 %iv, -1
+ %ec = icmp ugt i32 %iv, %lim
+ br i1 %ec, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; As above, signed.
+define void @sgt_guarded_bound_is_add(i32 %x, i32 %a, i32 %b) {
+; CHECK-LABEL: 'sgt_guarded_bound_is_add'
+; CHECK-NEXT: Determining loop execution counts for: @sgt_guarded_bound_is_add
+; CHECK-NEXT: Loop %loop: backedge-taken count is (-1 + (zext i8 (trunc i32 %x to i8) to i32) + (-1 * (%a + %b)))
+; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i32 -2147483394
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is (-1 + (zext i8 (trunc i32 %x to i8) to i32) + (-1 * (%a + %b)))
+; CHECK-NEXT: Loop %loop: Trip multiple is 1
+;
+entry:
+ %n = and i32 %x, 255
+ %lim = add i32 %a, %b
+ %start = add nsw i32 %n, -1
+ %guard = icmp sgt i32 %n, %lim
+ br i1 %guard, label %loop, label %exit
+
+loop:
+ %iv = phi i32 [ %start, %entry ], [ %iv.next, %loop ]
+ %iv.next = add nsw i32 %iv, -1
+ %ec = icmp sgt i32 %iv, %lim
+ br i1 %ec, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+define void @ugt_guarded_bound_is_add_with_constant(i32 %x, i32 %a, i32 %b) {
+; CHECK-LABEL: 'ugt_guarded_bound_is_add_with_constant'
+; CHECK-NEXT: Determining loop execution counts for: @ugt_guarded_bound_is_add_with_constant
+; CHECK-NEXT: Loop %loop: backedge-taken count is (-8 + (zext i8 (trunc i32 %x to i8) to i32) + (-1 * %a) + (-1 * %b))
+; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i32 -1
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is (-8 + (zext i8 (trunc i32 %x to i8) to i32) + (-1 * %a) + (-1 * %b))
+; CHECK-NEXT: Loop %loop: Trip multiple is 1
+;
+entry:
+ %n = and i32 %x, 255
+ %ab = add i32 %a, %b
+ %lim = add i32 %ab, 7
+ %start = add nsw i32 %n, -1
+ %guard = icmp ugt i32 %n, %lim
+ br i1 %guard, label %loop, label %exit
+
+loop:
+ %iv = phi i32 [ %start, %entry ], [ %iv.next, %loop ]
+ %iv.next = add i32 %iv, -1
+ %ec = icmp ugt i32 %iv, %lim
+ br i1 %ec, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; The entry guard proves the decrement is positive.
+define void @sgt_variable_stride_guard(i32 %start, i32 %bound, i32 %stride) {
+; CHECK-LABEL: 'sgt_variable_stride_guard'
+; CHECK-NEXT: Determining loop execution counts for: @sgt_variable_stride_guard
+; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+;
+entry:
+ %guard = icmp sgt i32 %stride, 0
+ br i1 %guard, label %ph, label %exit
+
+ph:
+ %step = sub i32 0, %stride
+ br label %loop
+
+loop:
+ %iv = phi i32 [ %start, %ph ], [ %iv.next, %loop ]
+ %iv.next = add nsw i32 %iv, %step
+ %ec = icmp sgt i32 %iv, %bound
+ br i1 %ec, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; Without a guard, the decrement may be zero or negative.
+define void @sgt_variable_stride_no_guard(i32 %start, i32 %bound, i32 %stride) {
+; CHECK-LABEL: 'sgt_variable_stride_no_guard'
+; CHECK-NEXT: Determining loop execution counts for: @sgt_variable_stride_no_guard
+; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+;
+entry:
+ br label %ph
+
+ph:
+ %step = sub i32 0, %stride
+ br label %loop
+
+loop:
+ %iv = phi i32 [ %start, %ph ], [ %iv.next, %loop ]
+ %iv.next = add nsw i32 %iv, %step
+ %ec = icmp sgt i32 %iv, %bound
+ br i1 %ec, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; A nonnegative decrement may still be zero, so the loop may not terminate.
+define void @sgt_variable_stride_nonnegative_guard(i32 %start, i32 %bound, i32 %stride) {
+; CHECK-LABEL: 'sgt_variable_stride_nonnegative_guard'
+; CHECK-NEXT: Determining loop execution counts for: @sgt_variable_stride_nonnegative_guard
+; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+;
+entry:
+ %guard = icmp sge i32 %stride, 0
+ br i1 %guard, label %ph, label %exit
+
+ph:
+ %step = sub i32 0, %stride
+ br label %loop
+
+loop:
+ %iv = phi i32 [ %start, %ph ], [ %iv.next, %loop ]
+ %iv.next = add nsw i32 %iv, %step
+ %ec = icmp sgt i32 %iv, %bound
+ br i1 %ec, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+; Negating INT_MIN wraps to INT_MIN; it does not give a positive stride.
+define void @sgt_variable_stride_min_guard(i32 %start, i32 %bound, i32 %stride) {
+; CHECK-LABEL: 'sgt_variable_stride_min_guard'
+; CHECK-NEXT: Determining loop execution counts for: @sgt_variable_stride_min_guard
+; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+;
+entry:
+ %guard = icmp eq i32 %stride, -2147483648
+ br i1 %guard, label %ph, label %exit
+
+ph:
+ %step = sub i32 0, %stride
+ br label %loop
+
+loop:
+ %iv = phi i32 [ %start, %ph ], [ %iv.next, %loop ]
+ %iv.next = add nsw i32 %iv, %step
+ %ec = icmp sgt i32 %iv, %bound
+ br i1 %ec, label %loop, label %exit
+
+exit:
+ ret void
+}
>From 45283e511805ccd5c48b6b44c10672bdc931f6a1 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 10 Sep 2026 15:54:07 +0100
Subject: [PATCH 2/2] [SCEV] Use howManyLessThans to implement
howManyGreaterThans.
Update howManyLessThans to support inverting the analyzed comparison and
analyze LHS > RHS as ~LHS < ~RHS.
howManyLessThans should now handle all cases howManyGreaterThans did
(and a few more, see improvements in
https://github.com/dtcxzyw/llvm-opt-benchmark-nightly/pull/1443).
howManyLessThans now accepts an Inverted argument, which analyzes ~LHS <
~RHS, without materializing the inverted operands explicitly, which can
pessimize results.
I tried to update all code paths where this is possible without too many
changes. A few code paths need bigger changes, and are skipped for now.
---
llvm/include/llvm/Analysis/ScalarEvolution.h | 10 +-
llvm/lib/Analysis/ScalarEvolution.cpp | 207 +++++++-----------
.../exit-count-greater-than.ll | 32 +--
.../exit-value-nowrap-flags.ll | 12 +-
.../Analysis/ScalarEvolution/trip-count13.ll | 4 +-
.../arm_cmplx_dot_prod_f32.ll | 9 +-
llvm/test/CodeGen/Thumb2/mve-pipelineloops.ll | 14 +-
7 files changed, 112 insertions(+), 176 deletions(-)
diff --git a/llvm/include/llvm/Analysis/ScalarEvolution.h b/llvm/include/llvm/Analysis/ScalarEvolution.h
index 4d9f0247ef640..f2275513a7db8 100644
--- a/llvm/include/llvm/Analysis/ScalarEvolution.h
+++ b/llvm/include/llvm/Analysis/ScalarEvolution.h
@@ -2229,7 +2229,9 @@ class ScalarEvolution {
/// less-than comparison will execute. If not computable, return
/// CouldNotCompute.
///
- /// \p isSigned specifies whether the less-than is signed.
+ /// \p IsSigned specifies whether the less-than is signed.
+ ///
+ /// If \p Invert is set, analyze "LHS > RHS" as "~LHS < ~RHS".
///
/// \p ControlsOnlyExit is true when the LHS < RHS condition directly controls
/// the branch (loops exits only if condition is true). In this case, we can
@@ -2238,13 +2240,9 @@ class ScalarEvolution {
/// If \p AllowPredicates is set, this call will try to use a minimal set of
/// SCEV predicates in order to return an exact answer.
ExitLimit howManyLessThans(const SCEV *LHS, const SCEV *RHS, const Loop *L,
- bool isSigned, bool ControlsOnlyExit,
+ bool IsSigned, bool Invert, bool ControlsOnlyExit,
bool AllowPredicates = false);
- ExitLimit howManyGreaterThans(const SCEV *LHS, const SCEV *RHS, const Loop *L,
- bool isSigned, bool IsSubExpr,
- bool AllowPredicates = false);
-
/// Return a predecessor of BB (which may not be an immediate predecessor)
/// which has exactly one successor from which BB is reachable, or null if
/// no such block is found.
diff --git a/llvm/lib/Analysis/ScalarEvolution.cpp b/llvm/lib/Analysis/ScalarEvolution.cpp
index 5b8e30985ee7f..f70207dc05d33 100644
--- a/llvm/lib/Analysis/ScalarEvolution.cpp
+++ b/llvm/lib/Analysis/ScalarEvolution.cpp
@@ -9525,8 +9525,8 @@ ScalarEvolution::ExitLimit ScalarEvolution::computeExitLimitFromICmp(
case ICmpInst::ICMP_SLT:
case ICmpInst::ICMP_ULT: { // while (X < Y)
bool IsSigned = ICmpInst::isSigned(Pred);
- ExitLimit EL = howManyLessThans(LHS, RHS, L, IsSigned, ControlsOnlyExit,
- AllowPredicates);
+ ExitLimit EL = howManyLessThans(LHS, RHS, L, IsSigned, /*Invert=*/false,
+ ControlsOnlyExit, AllowPredicates);
if (EL.hasAnyInfo())
return EL;
break;
@@ -9542,9 +9542,10 @@ ScalarEvolution::ExitLimit ScalarEvolution::computeExitLimitFromICmp(
[[fallthrough]];
case ICmpInst::ICMP_SGT:
case ICmpInst::ICMP_UGT: { // while (X > Y)
+ // "X > Y" is analyzed as the equivalent "~X < ~Y".
bool IsSigned = ICmpInst::isSigned(Pred);
- ExitLimit EL = howManyGreaterThans(LHS, RHS, L, IsSigned, ControlsOnlyExit,
- AllowPredicates);
+ ExitLimit EL = howManyLessThans(LHS, RHS, L, IsSigned, /*Invert=*/true,
+ ControlsOnlyExit, AllowPredicates);
if (EL.hasAnyInfo())
return EL;
break;
@@ -13383,13 +13384,18 @@ ScalarEvolution::computeMaxBECountForLT(const SCEV *Start, const SCEV *Stride,
ScalarEvolution::ExitLimit
ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
- const Loop *L, bool IsSigned,
+ const Loop *L, bool IsSigned, bool Invert,
bool ControlsOnlyExit, bool AllowPredicates) {
SmallVector<const SCEVPredicate *> Predicates;
+ // FIXME: Extend the non-invariant RHS analysis to greater-than comparisons.
+ if (Invert && !isLoopInvariant(RHS, L))
+ return getCouldNotCompute();
+
const SCEVAddRecExpr *IV = dyn_cast<SCEVAddRecExpr>(LHS);
bool PredicatedIV = false;
- if (!IV) {
+ // FIXME: Generalize the NUW inference below to decreasing IVs.
+ if (!IV && !Invert) {
if (auto *ZExt = dyn_cast<SCEVZeroExtendExpr>(LHS)) {
const SCEVAddRecExpr *AR = dyn_cast<SCEVAddRecExpr>(ZExt->getOperand());
if (AR && AR->getLoop() == L && AR->isAffine()) {
@@ -13439,7 +13445,6 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
}
}
-
if (!IV && AllowPredicates) {
// Try to make this an AddRec using runtime tests, in the first X
// iterations of this loop, where X is the SCEV expression found by the
@@ -13464,12 +13469,19 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
// exiting instruction we're analyzing would trigger UB.
auto WrapType = IsSigned ? SCEV::FlagNSW : SCEV::FlagNUW;
bool NoWrap = ControlsOnlyExit && any(IV->getNoWrapFlags(WrapType));
+ // Reverse the ordering for greater-than comparisons.
ICmpInst::Predicate Cond = IsSigned ? ICmpInst::ICMP_SLT : ICmpInst::ICMP_ULT;
+ if (Invert)
+ Cond = ICmpInst::getSwappedPredicate(Cond);
+ // The step of ~IV is the negated step of IV.
const SCEV *Stride = IV->getStepRecurrence(*this);
+ if (Invert)
+ Stride = getNegativeSCEV(Stride);
const SCEV *GuardedStride = Stride;
- // Whether the IV may reach the maximum value before the exit is taken.
+ // Whether the IV may reach the maximum (or minimum if inverted) value
+ // before the exit is taken.
bool IVMayOverflow = true;
bool PositiveStride = isKnownPositive(Stride);
@@ -13486,6 +13498,10 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
// Avoid negative or zero stride values.
if (!PositiveStride) {
+ // FIXME: Generalize the unknown-stride analysis to decreasing IVs.
+ if (Invert)
+ return getCouldNotCompute();
+
// We can compute the correct backedge taken count for loops with unknown
// strides if we can prove that the loop is not an infinite loop with side
// effects. Here's the loop structure we are trying to handle -
@@ -13570,7 +13586,7 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
} else {
// Avoid proven overflow cases: this will ensure that the backedge taken
// count will not generate any unsigned overflow.
- IVMayOverflow = canIVOverflowOnLT(RHS, GuardedStride, IsSigned);
+ IVMayOverflow = canIVOverflowOnLT(RHS, GuardedStride, IsSigned, Invert);
if (IVMayOverflow && !NoWrap)
return getCouldNotCompute();
}
@@ -13606,6 +13622,7 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
const SCEV *End = nullptr, *BECount = getCouldNotCompute(),
*BECountIfBackedgeTaken = getCouldNotCompute();
if (!isLoopInvariant(RHS, L)) {
+ assert(!Invert && "RHS must be loop-invariant for Invert");
const auto *RHSAddRec = dyn_cast<SCEVAddRecExpr>(RHS);
if (PositiveStride && RHSAddRec != nullptr && RHSAddRec->getLoop() == L &&
any(RHSAddRec->getNoWrapFlags())) {
@@ -13652,10 +13669,11 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
// Let End = max(RHS,Start). We use the expression (End-Start)/Stride to
// describe the backedge count: if the backedge is taken at least once then
// End is RHS, and if not End is Start so we get a backedge count of zero.
+ // Inverted, End is min(RHS, Start).
//
// AddingStrideMinusOneMayOverflow has the following preconditions:
//
- // 1. If IsSigned, Start <=s End; otherwise, Start <=u End
+ // 1. Start <= End, signed if IsSigned (inverted: End <= Start)
// 2. The index variable doesn't overflow.
//
// Therefore, we know N exists such that
@@ -13713,9 +13731,10 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
// Just rewrite steps before "End - Start <= Stride * N <= UMAX"
// to use signed max instead of unsigned max. Note that we're
// trying to prove a lack of unsigned overflow in either case.
+ // Inverted: "Start - End <= Stride * N <= Start - MIN <= UMAX", same.
return false;
}
- if (Start == Stride || Start == getMinusSCEV(Stride, One)) {
+ if (!Invert && (Start == Stride || Start == getMinusSCEV(Stride, One))) {
// If Start is equal to Stride, (End - Start) + (Stride - 1) == End
// - 1. If !IsSigned, 0 <u Stride == Start <=u End; so 0 <u End - 1
// <u End. If IsSigned, 0 <s Stride == Start <=s End; so 0 <s End -
@@ -13723,28 +13742,42 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
//
// If Start is equal to Stride - 1, (End - Start) + Stride - 1 ==
// End.
+ //
+ // Both need Start to be the smaller value, so neither applies inverted.
return false;
}
return true;
}();
- auto *OrigStartMinusStride = getMinusSCEV(OrigStart, Stride);
- assert(isAvailableAtLoopEntry(OrigStartMinusStride, L) && "Must be!");
+ // If inverted, the analyzed values are complements: "~V - Offset" is "~(V +
+ // Offset)" and "~To - ~From" is "From - To".
+ auto StepBack = [&](const SCEV *V, const SCEV *Offset) -> const SCEV * {
+ if (Invert)
+ return getAddExpr(V, Offset);
+ return getMinusSCEV(V, Offset);
+ };
+ auto Distance = [&](const SCEV *From, const SCEV *To) {
+ return Invert ? getMinusSCEV(From, To) : getMinusSCEV(To, From);
+ };
+
+ const SCEV *OrigPrevStart = StepBack(OrigStart, Stride);
+ assert(isAvailableAtLoopEntry(OrigPrevStart, L) && "Must be!");
assert(isAvailableAtLoopEntry(OrigStart, L) && "Must be!");
assert(isAvailableAtLoopEntry(OrigRHS, L) && "Must be!");
// Can we prove Start - Stride < RHS, and either Start - Stride < Start or
// (via !AddingStrideMinusOneMayOverflow) that (RHS - Start) + (Stride - 1)
// does not overflow?
if ((!AddingStrideMinusOneMayOverflow ||
- isLoopEntryGuardedByCond(L, Cond, OrigStartMinusStride, OrigStart)) &&
- isLoopEntryGuardedByCond(L, Cond, OrigStartMinusStride, OrigRHS)) {
+ isLoopEntryGuardedByCond(L, Cond, OrigPrevStart, OrigStart)) &&
+ isLoopEntryGuardedByCond(L, Cond, OrigPrevStart, OrigRHS)) {
// In this case, we can use a refined formula for computing backedge
// taken count. The general formula remains:
// "End-Start /uceiling Stride"
// We want to use the alternate formula:
// "((RHS - 1) - (Start - Stride)) /u Stride"
// Let's do a quick case analysis to show these are equivalent under
- // our preconditions.
+ // our preconditions. When inverted, the proof uses complemented Start,
+ // RHS and End; Stride remains positive.
// * For RHS <= Start (End is Start), the backedge-taken count must be
// zero. Together with the precondition "Start - Stride < RHS", we have
// "Start - Stride < RHS <= Start". Subtracting Start - Stride from
@@ -13766,19 +13799,29 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
// "End" is "RHS", as "RHS > Start", so this is the reassociated
// numerator. Neither sub-term wraps unsigned: "RHS - Start"
// due to "RHS > Start", and "Stride - 1", as Stride is non-zero.
- const SCEV *MinusOne = getMinusOne(Stride->getType());
const SCEV *Numerator =
- getMinusSCEV(getAddExpr(RHS, MinusOne), getMinusSCEV(Start, Stride));
+ getMinusSCEV(Distance(StepBack(Start, Stride), RHS), One);
BECount = getUDivExpr(Numerator, Stride);
}
if (isa<SCEVCouldNotCompute>(BECount)) {
- auto canProveRHSGreaterThanEqualStart = [&]() {
+ auto canProveRHSIsAtOrBeyondStart = [&]() {
+ // Inverted, the claim is "Start >= RHS". Reverse the comparisons below
+ // by swapping their operands rather than their predicates:
+ // isLoopEntryGuardedByCond is sensitive to operand order and loses the
+ // proof if the IV bound moves to the other side.
+ auto SwapIfInverted = [&](const SCEV *A, const SCEV *B) {
+ return Invert ? std::pair(B, A) : std::pair(A, B);
+ };
+
auto CondGE = IsSigned ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE;
const SCEV *GuardedRHS = applyLoopGuards(OrigRHS, L);
const SCEV *GuardedStart = applyLoopGuards(OrigStart, L);
+ if (Invert)
+ std::swap(GuardedRHS, GuardedStart);
- if (isLoopEntryGuardedByCond(L, CondGE, OrigRHS, OrigStart) ||
+ auto [GELHS, GERHS] = SwapIfInverted(OrigRHS, OrigStart);
+ if (isLoopEntryGuardedByCond(L, CondGE, GELHS, GERHS) ||
isKnownPredicate(CondGE, GuardedRHS, GuardedStart))
return true;
@@ -13792,14 +13835,13 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
//
// FIXME: Should isLoopEntryGuardedByCond do this for us?
auto CondGT = IsSigned ? ICmpInst::ICMP_SGT : ICmpInst::ICMP_UGT;
- const SCEV *StartMinusOne =
- getAddExpr(OrigStart, getMinusOne(OrigStart->getType()));
- return isLoopEntryGuardedByCond(L, CondGT, OrigRHS, StartMinusOne);
+ auto [GTLHS, GTRHS] = SwapIfInverted(OrigRHS, StepBack(OrigStart, One));
+ return isLoopEntryGuardedByCond(L, CondGT, GTLHS, GTRHS);
};
// If we know that RHS >= Start in the context of loop, then we know
// that max(RHS, Start) = RHS at this point.
- if (canProveRHSGreaterThanEqualStart()) {
+ if (canProveRHSIsAtOrBeyondStart()) {
End = RHS;
} else {
// If RHS < Start, the backedge will be taken zero times. So in
@@ -13810,15 +13852,19 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
// We convert it to the following to make it more convenient for SCEV:
//
// ceil(max(RHS, Start) - Start) / Stride
- End = IsSigned ? getSMaxExpr(RHS, Start) : getUMaxExpr(RHS, Start);
+ //
+ // Inverted, this is ceil(Start - min(RHS, Start)) / Stride.
+ if (Invert)
+ End = IsSigned ? getSMinExpr(RHS, Start) : getUMinExpr(RHS, Start);
+ else
+ End = IsSigned ? getSMaxExpr(RHS, Start) : getUMaxExpr(RHS, Start);
// See what would happen if we assume the backedge is taken. This is
// used to compute MaxBECount.
- BECountIfBackedgeTaken =
- getUDivCeilSCEV(getMinusSCEV(RHS, Start), Stride);
+ BECountIfBackedgeTaken = getUDivCeilSCEV(Distance(Start, RHS), Stride);
}
- const SCEV *Delta = getMinusSCEV(End, Start);
+ const SCEV *Delta = Distance(Start, End);
if (!AddingStrideMinusOneMayOverflow) {
// floor((D + (S - 1)) / S)
// We prefer this formulation if it's legal because it's fewer
@@ -13838,7 +13884,7 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
} else {
ConstantMaxBECount = computeMaxBECountForLT(
Start, Stride, RHS, getTypeSizeInBits(LHS->getType()), IsSigned,
- /*Invert=*/false);
+ Invert);
// If we know exactly how many times the backedge will be taken if it's
// taken at least once, then the backedge count will either be that or
// zero. If that count exceeds the range-based bound, the backedge can
@@ -13865,109 +13911,6 @@ ScalarEvolution::howManyLessThans(const SCEV *LHS, const SCEV *RHS,
Predicates);
}
-ScalarEvolution::ExitLimit ScalarEvolution::howManyGreaterThans(
- const SCEV *LHS, const SCEV *RHS, const Loop *L, bool IsSigned,
- bool ControlsOnlyExit, bool AllowPredicates) {
- SmallVector<const SCEVPredicate *> Predicates;
- // We handle only IV > Invariant
- if (!isLoopInvariant(RHS, L))
- return getCouldNotCompute();
-
- const SCEVAddRecExpr *IV = dyn_cast<SCEVAddRecExpr>(LHS);
- if (!IV && AllowPredicates)
- // Try to make this an AddRec using runtime tests, in the first X
- // iterations of this loop, where X is the SCEV expression found by the
- // algorithm below.
- IV = convertSCEVToAddRecWithPredicates(LHS, L, Predicates);
-
- // Avoid weird loops
- if (!IV || IV->getLoop() != L || !IV->isAffine())
- return getCouldNotCompute();
-
- auto WrapType = IsSigned ? SCEV::FlagNSW : SCEV::FlagNUW;
- bool NoWrap = ControlsOnlyExit && any(IV->getNoWrapFlags(WrapType));
- ICmpInst::Predicate Cond = IsSigned ? ICmpInst::ICMP_SGT : ICmpInst::ICMP_UGT;
-
- const SCEV *Stride = getNegativeSCEV(IV->getStepRecurrence(*this));
-
- // Avoid negative or zero stride values
- if (!isKnownPositive(Stride))
- return getCouldNotCompute();
-
- // Avoid proven overflow cases: this will ensure that the backedge taken count
- // will not generate any unsigned overflow. Relaxed no-overflow conditions
- // exploit no-wrap flags, allowing to optimize in presence of undefined
- // behaviors like the case of C language.
- bool MayAddOverflow = false;
- const SCEV *Start = IV->getStart();
- const SCEV *End = RHS;
- if (!Stride->isOne() &&
- canIVOverflowOnLT(RHS, Stride, IsSigned, /*Invert=*/true)) {
- if (!NoWrap)
- return getCouldNotCompute();
- MayAddOverflow = true;
- }
-
- if (!isLoopEntryGuardedByCond(L, Cond, getAddExpr(Start, Stride), RHS)) {
- // If we know that Start >= RHS in the context of loop, then we know that
- // min(RHS, Start) = RHS at this point.
- if (isLoopEntryGuardedByCond(
- L, IsSigned ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE, Start, RHS))
- End = RHS;
- else
- End = IsSigned ? getSMinExpr(RHS, Start) : getUMinExpr(RHS, Start);
- }
-
- if (Start->getType()->isPointerTy()) {
- assert(End->getType()->isPointerTy() && RHS->getType()->isPointerTy() &&
- "Start, End and RHS all must be pointers");
- Start = getPtrToAddrExpr(Start);
- if (isa<SCEVCouldNotCompute>(Start))
- return Start;
-
- End = getPtrToAddrExpr(End);
- if (isa<SCEVCouldNotCompute>(End))
- return End;
-
- RHS = getPtrToAddrExpr(RHS);
- if (isa<SCEVCouldNotCompute>(RHS))
- return RHS;
- }
-
- const SCEV *Delta = getMinusSCEV(Start, End);
- const SCEV *BECount;
- if (MayAddOverflow) {
- // The ceiling division instead needs Start >= End, so that (Start - End) is
- // the exact unsigned distance between them.
- if (!isLoopEntryGuardedByCond(
- L, IsSigned ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE, Start, End))
- return getCouldNotCompute();
- BECount = getUDivCeilSCEV(Delta, Stride);
- } else {
- // Compute ((Start - End) + (Stride - 1)) / Stride, if the IV cannot
- // overflow as it requires fewer operations.
- const SCEV *One = getOne(Stride->getType());
- BECount = getUDivExpr(getAddExpr(Delta, getMinusSCEV(Stride, One)), Stride);
- }
-
- // "IV > RHS" is analyzed as the equivalent "~IV < ~RHS"; Stride is already
- // the negated step.
- const SCEV *ConstantMaxBECount =
- isa<SCEVConstant>(BECount)
- ? BECount
- : computeMaxBECountForLT(Start, Stride, RHS,
- getTypeSizeInBits(LHS->getType()), IsSigned,
- /*Invert=*/true);
-
- if (isa<SCEVCouldNotCompute>(ConstantMaxBECount))
- ConstantMaxBECount = BECount;
- const SCEV *SymbolicMaxBECount =
- isa<SCEVCouldNotCompute>(BECount) ? ConstantMaxBECount : BECount;
-
- return ExitLimit(BECount, ConstantMaxBECount, SymbolicMaxBECount, false,
- Predicates);
-}
-
const SCEV *SCEVAddRecExpr::getNumIterationsInRange(const ConstantRange &Range,
ScalarEvolution &SE) const {
if (Range.isFullSet()) // Infinite loop.
diff --git a/llvm/test/Analysis/ScalarEvolution/exit-count-greater-than.ll b/llvm/test/Analysis/ScalarEvolution/exit-count-greater-than.ll
index 5cdad4418d45b..6ff3d51714c55 100644
--- a/llvm/test/Analysis/ScalarEvolution/exit-count-greater-than.ll
+++ b/llvm/test/Analysis/ScalarEvolution/exit-count-greater-than.ll
@@ -86,9 +86,10 @@ exit:
define i32 @sgt_stride_4_variable_bound(i32 %n, i32 %m) {
; CHECK-LABEL: 'sgt_stride_4_variable_bound'
; CHECK-NEXT: Determining loop execution counts for: @sgt_stride_4_variable_bound
-; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
-; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
-; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+; CHECK-NEXT: Loop %loop: backedge-taken count is ((3 + (-1 * %m) + %n) /u 4)
+; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i32 1073741823
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is ((3 + (-1 * %m) + %n) /u 4)
+; CHECK-NEXT: Loop %loop: Trip multiple is 1
;
entry:
%add = add nsw i32 %n, 4
@@ -115,9 +116,10 @@ ret:
define i32 @ugt_stride_4_variable_bound(i32 %n, i32 %m) {
; CHECK-LABEL: 'ugt_stride_4_variable_bound'
; CHECK-NEXT: Determining loop execution counts for: @ugt_stride_4_variable_bound
-; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
-; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
-; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+; CHECK-NEXT: Loop %loop: backedge-taken count is ((3 + (-1 * %m) + %n) /u 4)
+; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i32 1073741823
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is ((3 + (-1 * %m) + %n) /u 4)
+; CHECK-NEXT: Loop %loop: Trip multiple is 1
;
entry:
%add = add nuw i32 %n, 4
@@ -145,9 +147,10 @@ ret:
define i32 @sgt_stride_3_variable_bound(i32 %n, i32 %m) {
; CHECK-LABEL: 'sgt_stride_3_variable_bound'
; CHECK-NEXT: Determining loop execution counts for: @sgt_stride_3_variable_bound
-; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
-; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
-; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+; CHECK-NEXT: Loop %loop: backedge-taken count is ((((-1 * (1 umin ((-1 * (%n smin %m)) + %n)))<nuw><nsw> + (-1 * (%n smin %m)) + %n) /u 3) + (1 umin ((-1 * (%n smin %m)) + %n)))
+; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i32 1431655765
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is ((((-1 * (1 umin ((-1 * (%n smin %m)) + %n)))<nuw><nsw> + (-1 * (%n smin %m)) + %n) /u 3) + (1 umin ((-1 * (%n smin %m)) + %n)))
+; CHECK-NEXT: Loop %loop: Trip multiple is 1
;
entry:
%add = add nsw i32 %n, 3
@@ -170,9 +173,9 @@ ret:
define void @sgt_stride_4_no_overflow(i32 %n) {
; CHECK-LABEL: 'sgt_stride_4_no_overflow'
; CHECK-NEXT: Determining loop execution counts for: @sgt_stride_4_no_overflow
-; CHECK-NEXT: Loop %loop: backedge-taken count is ((3 + %n) /u 4)
+; CHECK-NEXT: Loop %loop: backedge-taken count is ((3 + %n)<nsw> /u 4)
; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i32 536870912
-; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is ((3 + %n) /u 4)
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is ((3 + %n)<nsw> /u 4)
; CHECK-NEXT: Loop %loop: Trip multiple is 1
;
entry:
@@ -371,9 +374,10 @@ exit:
define void @sgt_variable_stride_guard(i32 %start, i32 %bound, i32 %stride) {
; CHECK-LABEL: 'sgt_variable_stride_guard'
; CHECK-NEXT: Determining loop execution counts for: @sgt_variable_stride_guard
-; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
-; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
-; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+; CHECK-NEXT: Loop %loop: backedge-taken count is ((((-1 * (1 umin ((-1 * (%start smin %bound)) + %start)))<nuw><nsw> + (-1 * (%start smin %bound)) + %start) /u (1 umax %stride)) + (1 umin ((-1 * (%start smin %bound)) + %start)))
+; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i32 -1
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is ((((-1 * (1 umin ((-1 * (%start smin %bound)) + %start)))<nuw><nsw> + (-1 * (%start smin %bound)) + %start) /u (1 umax %stride)) + (1 umin ((-1 * (%start smin %bound)) + %start)))
+; CHECK-NEXT: Loop %loop: Trip multiple is 1
;
entry:
%guard = icmp sgt i32 %stride, 0
diff --git a/llvm/test/Analysis/ScalarEvolution/exit-value-nowrap-flags.ll b/llvm/test/Analysis/ScalarEvolution/exit-value-nowrap-flags.ll
index bce3e717a7f47..a34532bcd95eb 100644
--- a/llvm/test/Analysis/ScalarEvolution/exit-value-nowrap-flags.ll
+++ b/llvm/test/Analysis/ScalarEvolution/exit-value-nowrap-flags.ll
@@ -18,8 +18,8 @@ define void @dec_to_start_of_nuw_addrec(i64 %start) {
; CHECK-NEXT: --> {{\{\{}}(-1 + %start),+,1}<nw><%up>,+,-1}<nw><%down> U: full-set S: full-set --> {(8 + %start),+,-1}<nw><%down> U: full-set S: full-set Exits: ((8 + %start) umin %start) LoopDispositions: { %down: Computable }
; CHECK-NEXT: Determining loop execution counts for: @dec_to_start_of_nuw_addrec
; CHECK-NEXT: Loop %down: backedge-taken count is (8 + (-1 * ((8 + %start) umin %start)) + %start)
-; CHECK-NEXT: Loop %down: constant max backedge-taken count is i64 8
-; CHECK-NEXT: Loop %down: symbolic max backedge-taken count is (8 + (-1 * ((8 + %start) umin %start)) + %start)
+; CHECK-NEXT: Loop %down: constant max backedge-taken count is i64 8, actual taken count either this or zero.
+; CHECK-NEXT: Loop %down: symbolic max backedge-taken count is (8 + (-1 * ((8 + %start) umin %start)) + %start), actual taken count either this or zero.
; CHECK-NEXT: Loop %down: Trip multiple is 1
; CHECK-NEXT: Loop %up: backedge-taken count is i32 9
; CHECK-NEXT: Loop %up: constant max backedge-taken count is i32 9
@@ -64,8 +64,8 @@ define void @dec_to_start_of_nuw_ptr_addrec(ptr %start) {
; CHECK-NEXT: --> {{\{\{}}(-1 + %start),+,1}<nw><%up>,+,-1}<nw><%down> U: full-set S: full-set --> {(8 + %start),+,-1}<nw><%down> U: full-set S: full-set Exits: ((-1 * (ptrtoaddr ptr %start to i64)) + ((8 + (ptrtoaddr ptr %start to i64)) umin (ptrtoaddr ptr %start to i64)) + %start) LoopDispositions: { %down: Computable }
; CHECK-NEXT: Determining loop execution counts for: @dec_to_start_of_nuw_ptr_addrec
; CHECK-NEXT: Loop %down: backedge-taken count is (8 + (-1 * ((8 + (ptrtoaddr ptr %start to i64)) umin (ptrtoaddr ptr %start to i64))) + (ptrtoaddr ptr %start to i64))
-; CHECK-NEXT: Loop %down: constant max backedge-taken count is i64 8
-; CHECK-NEXT: Loop %down: symbolic max backedge-taken count is (8 + (-1 * ((8 + (ptrtoaddr ptr %start to i64)) umin (ptrtoaddr ptr %start to i64))) + (ptrtoaddr ptr %start to i64))
+; CHECK-NEXT: Loop %down: constant max backedge-taken count is i64 8, actual taken count either this or zero.
+; CHECK-NEXT: Loop %down: symbolic max backedge-taken count is (8 + (-1 * ((8 + (ptrtoaddr ptr %start to i64)) umin (ptrtoaddr ptr %start to i64))) + (ptrtoaddr ptr %start to i64)), actual taken count either this or zero.
; CHECK-NEXT: Loop %down: Trip multiple is 1
; CHECK-NEXT: Loop %up: backedge-taken count is i32 9
; CHECK-NEXT: Loop %up: constant max backedge-taken count is i32 9
@@ -110,8 +110,8 @@ define void @dec_to_start_of_wrapping_addrec(i64 %start) {
; CHECK-NEXT: --> {{\{\{}}(-1 + %start),+,1}<nw><%up>,+,-1}<nw><%down> U: full-set S: full-set --> {(8 + %start),+,-1}<nw><%down> U: full-set S: full-set Exits: ((8 + %start) umin %start) LoopDispositions: { %down: Computable }
; CHECK-NEXT: Determining loop execution counts for: @dec_to_start_of_wrapping_addrec
; CHECK-NEXT: Loop %down: backedge-taken count is (8 + (-1 * ((8 + %start) umin %start)) + %start)
-; CHECK-NEXT: Loop %down: constant max backedge-taken count is i64 8
-; CHECK-NEXT: Loop %down: symbolic max backedge-taken count is (8 + (-1 * ((8 + %start) umin %start)) + %start)
+; CHECK-NEXT: Loop %down: constant max backedge-taken count is i64 8, actual taken count either this or zero.
+; CHECK-NEXT: Loop %down: symbolic max backedge-taken count is (8 + (-1 * ((8 + %start) umin %start)) + %start), actual taken count either this or zero.
; CHECK-NEXT: Loop %down: Trip multiple is 1
; CHECK-NEXT: Loop %up: backedge-taken count is i32 9
; CHECK-NEXT: Loop %up: constant max backedge-taken count is i32 9
diff --git a/llvm/test/Analysis/ScalarEvolution/trip-count13.ll b/llvm/test/Analysis/ScalarEvolution/trip-count13.ll
index 7b6229f127cd5..8f04aadd69072 100644
--- a/llvm/test/Analysis/ScalarEvolution/trip-count13.ll
+++ b/llvm/test/Analysis/ScalarEvolution/trip-count13.ll
@@ -106,8 +106,8 @@ define void @s_2(i8 %start) {
; CHECK-LABEL: 's_2'
; CHECK-NEXT: Determining loop execution counts for: @s_2
; CHECK-NEXT: Loop %loop: backedge-taken count is ((-1 * ((-100 + %start) smin %start)) + %start)
-; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i8 100
-; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is ((-1 * ((-100 + %start) smin %start)) + %start)
+; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i8 100, actual taken count either this or zero.
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is ((-1 * ((-100 + %start) smin %start)) + %start), actual taken count either this or zero.
; CHECK-NEXT: Loop %loop: Trip multiple is 1
;
entry:
diff --git a/llvm/test/CodeGen/Thumb2/LowOverheadLoops/arm_cmplx_dot_prod_f32.ll b/llvm/test/CodeGen/Thumb2/LowOverheadLoops/arm_cmplx_dot_prod_f32.ll
index a87d363fa61ee..71b07bd313336 100644
--- a/llvm/test/CodeGen/Thumb2/LowOverheadLoops/arm_cmplx_dot_prod_f32.ll
+++ b/llvm/test/CodeGen/Thumb2/LowOverheadLoops/arm_cmplx_dot_prod_f32.ll
@@ -12,16 +12,11 @@ define void @arm_cmplx_dot_prod_f32(ptr %pSrcA, ptr %pSrcB, i32 %numSamples, ptr
; CHECK-NEXT: cmp r2, #8
; CHECK-NEXT: blo .LBB0_6
; CHECK-NEXT: @ %bb.1: @ %while.body.preheader
-; CHECK-NEXT: lsrs r4, r2, #2
-; CHECK-NEXT: mov.w lr, #2
-; CHECK-NEXT: cmp r4, #2
+; CHECK-NEXT: mov.w r4, #-1
+; CHECK-NEXT: add.w lr, r4, r2, lsr #2
; CHECK-NEXT: vldrw.u32 q2, [r1], #32
; CHECK-NEXT: vldrw.u32 q1, [r0], #32
-; CHECK-NEXT: it lt
-; CHECK-NEXT: lsrlt.w lr, r2, #2
-; CHECK-NEXT: rsb r4, lr, r2, lsr #2
; CHECK-NEXT: vmov.i32 q0, #0x0
-; CHECK-NEXT: add.w lr, r4, #1
; CHECK-NEXT: .LBB0_2: @ %while.body
; CHECK-NEXT: @ =>This Inner Loop Header: Depth=1
; CHECK-NEXT: vcmla.f32 q0, q1, q2, #0
diff --git a/llvm/test/CodeGen/Thumb2/mve-pipelineloops.ll b/llvm/test/CodeGen/Thumb2/mve-pipelineloops.ll
index 43ed5eefbf4c7..5bfc96cf12def 100644
--- a/llvm/test/CodeGen/Thumb2/mve-pipelineloops.ll
+++ b/llvm/test/CodeGen/Thumb2/mve-pipelineloops.ll
@@ -10,18 +10,14 @@ define void @arm_cmplx_dot_prod_q15(ptr noundef %pSrcA, ptr noundef %pSrcB, i32
; CHECK-NEXT: cmp r2, #16
; CHECK-NEXT: blo .LBB0_5
; CHECK-NEXT: @ %bb.1: @ %while.body.preheader
-; CHECK-NEXT: movs r6, #2
-; CHECK-NEXT: lsrs r7, r2, #3
-; CHECK-NEXT: rsb r6, r6, r2, lsr #3
-; CHECK-NEXT: cmp r7, #2
-; CHECK-NEXT: mov.w r5, #0
-; CHECK-NEXT: csel r7, r6, r5, hs
-; CHECK-NEXT: add.w lr, r7, #1
-; CHECK-NEXT: mov r4, r5
+; CHECK-NEXT: mov.w r7, #-1
+; CHECK-NEXT: movs r5, #0
+; CHECK-NEXT: add.w lr, r7, r2, lsr #3
; CHECK-NEXT: vldrh.u16 q0, [r0], #32
+; CHECK-NEXT: mov r4, r5
; CHECK-NEXT: movs r7, #0
-; CHECK-NEXT: mov r8, r5
; CHECK-NEXT: vldrh.u16 q1, [r1], #32
+; CHECK-NEXT: mov r8, r5
; CHECK-NEXT: vmlsldava.s16 r4, r7, q0, q1
; CHECK-NEXT: vldrh.u16 q2, [r0, #-16]
; CHECK-NEXT: vmlaldavax.s16 r8, r5, q0, q1
More information about the llvm-commits
mailing list