[llvm] [polly] [SCEV] Introduce SDiv expressions (PR #216862)
Ramkumar Ramachandra via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 00:48:37 PDT 2026
https://github.com/artagnon updated https://github.com/llvm/llvm-project/pull/216862
>From f74267a604d96301bb7726257aefe9ddc0bb7874 Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Mon, 17 Aug 2026 21:31:32 +0100
Subject: [PATCH 1/4] [SCEV] Introduce SDiv expressions
Introduce a signed-division ScalarEvolution expression. The initial
patch has been kept simple, and should be easy to reason about: we have
generalized the getUDivExpr into a getDivExpr, and disabled any
non-trivial folds. We have also refrained from doing anything other a
straight-forward expansion in ScalarEvolutionExpander. As
ScalarEvolution currently treats signed-division as a SCEVUnknown,
introducing SDiv expressions should be a strict improvement.
---
llvm/include/llvm/Analysis/ScalarEvolution.h | 6 +-
.../llvm/Analysis/ScalarEvolutionDivision.h | 1 +
.../Analysis/ScalarEvolutionExpressions.h | 86 +++-
.../Analysis/ScalarEvolutionPatternMatch.h | 7 +
.../Utils/ScalarEvolutionExpander.h | 2 +
llvm/lib/Analysis/ScalarEvolution.cpp | 286 ++++++++-----
.../Utils/ScalarEvolutionExpander.cpp | 55 +--
.../add-expr-pointer-operand-sorting.ll | 4 +-
.../addrec-may-wrap-sdiv-canonicalize.ll | 402 ++++++++++++++++++
.../extract-highbits-sameconstmask.ll | 4 +-
.../extract-highbits-variablemask.ll | 4 +-
.../ScalarEvolution/flags-from-poison.ll | 4 +-
.../ScalarEvolution/implied-via-division.ll | 92 ++--
.../ScalarEvolution/mul-sdiv-folds.ll | 145 +++++++
.../ptrtoint-constantexpr-loop.ll | 6 +-
llvm/test/Analysis/ScalarEvolution/sdiv.ll | 249 +++++++++++
.../CodeGen/Thumb2/mve-float16regloops.ll | 178 ++++----
.../CodeGen/Thumb2/mve-float32regloops.ll | 184 ++++----
llvm/test/CodeGen/X86/optimize-max-0.ll | 256 +++++------
.../Attributor/IPConstantProp/PR16052.ll | 4 +-
.../LICM/update-scev-after-hoist.ll | 8 +-
.../trip-count-expansion-may-introduce-ub.ll | 52 ++-
polly/include/polly/Support/SCEVAffinator.h | 1 +
polly/lib/Support/SCEVAffinator.cpp | 19 +-
polly/lib/Support/SCEVValidator.cpp | 31 +-
polly/lib/Support/ScopHelper.cpp | 21 +-
polly/test/CodeGen/inner_scev_sdiv_2.ll | 6 +-
.../CodeGen/scop_expander_insert_point.ll | 7 +-
.../ScopInfo/nonaffine-buildMemoryAccess.ll | 8 +-
29 files changed, 1550 insertions(+), 578 deletions(-)
create mode 100644 llvm/test/Analysis/ScalarEvolution/addrec-may-wrap-sdiv-canonicalize.ll
create mode 100644 llvm/test/Analysis/ScalarEvolution/mul-sdiv-folds.ll
diff --git a/llvm/include/llvm/Analysis/ScalarEvolution.h b/llvm/include/llvm/Analysis/ScalarEvolution.h
index b162036ffb46d..80d19cd595dbc 100644
--- a/llvm/include/llvm/Analysis/ScalarEvolution.h
+++ b/llvm/include/llvm/Analysis/ScalarEvolution.h
@@ -791,8 +791,10 @@ class ScalarEvolution {
SmallVector<SCEVUse, 3> Ops = {Op0, Op1, Op2};
return getMulExpr(Ops, Flags, Depth);
}
+ LLVM_ABI const SCEV *getDivExpr(bool IsSigned, SCEVUse LHS, SCEVUse RHS);
LLVM_ABI const SCEV *getUDivExpr(SCEVUse LHS, SCEVUse RHS);
LLVM_ABI const SCEV *getUDivExactExpr(SCEVUse LHS, SCEVUse RHS);
+ LLVM_ABI const SCEV *getSDivExpr(SCEVUse LHS, SCEVUse RHS);
LLVM_ABI const SCEV *getURemExpr(SCEVUse LHS, SCEVUse RHS);
LLVM_ABI const SCEV *getAddRecExpr(SCEVUse Start, SCEVUse Step, const Loop *L,
SCEV::NoWrapFlags Flags);
@@ -2537,8 +2539,8 @@ class ScalarEvolution {
const SCEV *getOrCreateAddRecExpr(ArrayRef<SCEVUse> Ops, const Loop *L,
SCEV::NoWrapFlags Flags);
- // Get UDiv expression already created or create a new one.
- const SCEV *getOrCreateUDivExpr(SCEVUse LHS, SCEVUse RHS);
+ // Get SDiv/UDiv expression already created or create a new one.
+ const SCEV *getOrCreateDivExpr(bool IsSigned, SCEVUse LHS, SCEVUse RHS);
/// Return x if \p Val is f(x) where f is a 1-1 function.
const SCEV *stripInjectiveFunctions(const SCEV *Val) const;
diff --git a/llvm/include/llvm/Analysis/ScalarEvolutionDivision.h b/llvm/include/llvm/Analysis/ScalarEvolutionDivision.h
index da873bda9db59..db24467246d1a 100644
--- a/llvm/include/llvm/Analysis/ScalarEvolutionDivision.h
+++ b/llvm/include/llvm/Analysis/ScalarEvolutionDivision.h
@@ -51,6 +51,7 @@ struct SCEVDivision : public SCEVVisitor<SCEVDivision, void> {
void visitZeroExtendExpr(const SCEVZeroExtendExpr *Numerator) {}
void visitSignExtendExpr(const SCEVSignExtendExpr *Numerator) {}
void visitUDivExpr(const SCEVUDivExpr *Numerator) {}
+ void visitSDivExpr(const SCEVSDivExpr *Numerator) {}
void visitSMaxExpr(const SCEVSMaxExpr *Numerator) {}
void visitUMaxExpr(const SCEVUMaxExpr *Numerator) {}
void visitSMinExpr(const SCEVSMinExpr *Numerator) {}
diff --git a/llvm/include/llvm/Analysis/ScalarEvolutionExpressions.h b/llvm/include/llvm/Analysis/ScalarEvolutionExpressions.h
index 63c822d0c7c20..da698a7a23965 100644
--- a/llvm/include/llvm/Analysis/ScalarEvolutionExpressions.h
+++ b/llvm/include/llvm/Analysis/ScalarEvolutionExpressions.h
@@ -46,6 +46,7 @@ enum SCEVTypes : unsigned short {
scAddExpr,
scMulExpr,
scUDivExpr,
+ scSDivExpr,
scAddRecExpr,
scUMaxExpr,
scSMaxExpr,
@@ -293,32 +294,84 @@ class SCEVMulExpr : public SCEVCommutativeExpr {
static bool classof(const SCEVUse *U) { return classof(U->getPointer()); }
};
-/// This class represents a binary unsigned division operation.
-class SCEVUDivExpr : public SCEV {
+/// This abstract class holds the code common to binary signed and unsigned
+/// divisions.
+class SCEVDivExpr : public SCEV {
friend class ScalarEvolution;
+ friend class SCEVUDivExpr;
+ friend class SCEVSDivExpr;
std::array<SCEVUse, 2> Operands;
- SCEVUDivExpr(const FoldingSetNodeIDRef ID, SCEVUse lhs, SCEVUse rhs)
- : SCEV(ID, scUDivExpr, computeExpressionSize({lhs, rhs}),
- lhs->getType()) {
- Operands[0] = lhs;
- Operands[1] = rhs;
+ SCEVDivExpr(const FoldingSetNodeIDRef ID, enum SCEVTypes T, SCEVUse LHS,
+ SCEVUse RHS)
+ : SCEV(ID, T, computeExpressionSize({LHS, RHS}), LHS->getType()) {
+ Operands[0] = LHS;
+ Operands[1] = RHS;
}
+ virtual ~SCEVDivExpr() = default;
+
public:
SCEVUse getLHS() const { return Operands[0]; }
SCEVUse getRHS() const { return Operands[1]; }
size_t getNumOperands() const { return 2; }
- SCEVUse getOperand(unsigned i) const {
- assert((i == 0 || i == 1) && "Operand index out of range!");
- return i == 0 ? getLHS() : getRHS();
+ SCEVUse getOperand(unsigned I) const {
+ assert(I < getNumOperands() && "Operand index out of range!");
+ return Operands[I];
}
ArrayRef<SCEVUse> operands() const { return Operands; }
+ /// Methods for support type inquiry through isa, cast, and dyn_cast:
+ static bool classof(const SCEV *S) {
+ return is_contained({scUDivExpr, scSDivExpr}, S->getSCEVType());
+ }
+ virtual bool mayTriggerUB(ScalarEvolution &SE) const = 0;
+};
+
+/// This class represents a binary unsigned division operation.
+class SCEVUDivExpr : public SCEVDivExpr {
+ friend class ScalarEvolution;
+
+ SCEVUDivExpr(const FoldingSetNodeIDRef ID, SCEVUse LHS, SCEVUse RHS)
+ : SCEVDivExpr(ID, scUDivExpr, LHS, RHS) {}
+
+ virtual ~SCEVUDivExpr() = default;
+
+public:
/// Methods for support type inquiry through isa, cast, and dyn_cast:
static bool classof(const SCEV *S) { return S->getSCEVType() == scUDivExpr; }
+
+ virtual bool mayTriggerUB(ScalarEvolution &SE) const override {
+ return !SE.isKnownNonZero(getRHS()) ||
+ !ScalarEvolution::isGuaranteedNotToBePoison(getRHS());
+ }
+};
+
+/// This class represents a binary signed division operation.
+class SCEVSDivExpr : public SCEVDivExpr {
+ friend class ScalarEvolution;
+
+ SCEVSDivExpr(const FoldingSetNodeIDRef ID, SCEVUse LHS, SCEVUse RHS)
+ : SCEVDivExpr(ID, scSDivExpr, LHS, RHS) {}
+
+ virtual ~SCEVSDivExpr() = default;
+
+public:
+ /// Methods for support type inquiry through isa, cast, and dyn_cast:
+ static bool classof(const SCEV *S) { return S->getSCEVType() == scSDivExpr; }
+
+ /// Return true for degenerate cases of this expression.
+ virtual bool mayTriggerUB(ScalarEvolution &SE) const override {
+ return !SE.isKnownNonZero(getRHS()) ||
+ (SE.getSignedRangeMin(getLHS()).isMinSignedValue() &&
+ !SE.isKnownPredicate(
+ CmpInst::ICMP_NE, getRHS(),
+ SE.getConstant(getType(), -1, /*isSigned=*/true))) ||
+ !ScalarEvolution::isGuaranteedNotToBePoison(getLHS()) ||
+ !ScalarEvolution::isGuaranteedNotToBePoison(getRHS());
+ }
};
/// This node represents a polynomial recurrence on the trip count
@@ -609,6 +662,8 @@ template <typename SC, typename RetVal = void> struct SCEVVisitor {
return ((SC *)this)->visitMulExpr((const SCEVMulExpr *)S);
case scUDivExpr:
return ((SC *)this)->visitUDivExpr((const SCEVUDivExpr *)S);
+ case scSDivExpr:
+ return ((SC *)this)->visitSDivExpr((const SCEVSDivExpr *)S);
case scAddRecExpr:
return ((SC *)this)->visitAddRecExpr((const SCEVAddRecExpr *)S);
case scSMaxExpr:
@@ -663,6 +718,9 @@ template <typename SC, typename RetVal = void> struct SCEVUseVisitor {
case scUDivExpr:
return ((SC *)this)
->visitUDivExpr(cast<SCEVUseT<const SCEVUDivExpr *>>(S));
+ case scSDivExpr:
+ return ((SC *)this)
+ ->visitSDivExpr(cast<SCEVUseT<const SCEVSDivExpr *>>(S));
case scAddRecExpr:
return ((SC *)this)
->visitAddRecExpr(cast<SCEVUseT<const SCEVAddRecExpr *>>(S));
@@ -734,6 +792,7 @@ template <typename SV> class SCEVTraversal {
case scAddExpr:
case scMulExpr:
case scUDivExpr:
+ case scSDivExpr:
case scSMaxExpr:
case scUMaxExpr:
case scSMinExpr:
@@ -869,6 +928,13 @@ class SCEVRewriteVisitor : public SCEVVisitor<SC, const SCEV *> {
return !Changed ? Expr : SE.getUDivExpr(LHS, RHS);
}
+ const SCEV *visitSDivExpr(const SCEVSDivExpr *Expr) {
+ auto *LHS = ((SC *)this)->visit(Expr->getLHS());
+ auto *RHS = ((SC *)this)->visit(Expr->getRHS());
+ bool Changed = LHS != Expr->getLHS() || RHS != Expr->getRHS();
+ return !Changed ? Expr : SE.getSDivExpr(LHS, RHS);
+ }
+
const SCEV *visitAddRecExpr(const SCEVAddRecExpr *Expr) {
SmallVector<SCEVUse, 2> Operands;
bool Changed = false;
diff --git a/llvm/include/llvm/Analysis/ScalarEvolutionPatternMatch.h b/llvm/include/llvm/Analysis/ScalarEvolutionPatternMatch.h
index 0ed08989e483b..a334a16c0779a 100644
--- a/llvm/include/llvm/Analysis/ScalarEvolutionPatternMatch.h
+++ b/llvm/include/llvm/Analysis/ScalarEvolutionPatternMatch.h
@@ -257,6 +257,13 @@ m_scev_c_NUWMul(const Op0_t &Op0, const Op1_t &Op1) {
Op1);
}
+template <typename Op0_t, typename Op1_t>
+inline SCEVBinaryExpr_match<SCEVMulExpr, Op0_t, Op1_t, SCEV::FlagNSW, true>
+m_scev_c_NSWMul(const Op0_t &Op0, const Op1_t &Op1) {
+ return m_scev_Binary<SCEVMulExpr, Op0_t, Op1_t, SCEV::FlagNSW, true>(Op0,
+ Op1);
+}
+
template <typename Op0_t, typename Op1_t>
inline SCEVBinaryExpr_match<SCEVUDivExpr, Op0_t, Op1_t>
m_scev_UDiv(const Op0_t &Op0, const Op1_t &Op1) {
diff --git a/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h b/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h
index c98c0cb52fa9c..484c357bfe99d 100644
--- a/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h
+++ b/llvm/include/llvm/Transforms/Utils/ScalarEvolutionExpander.h
@@ -538,6 +538,8 @@ class SCEVExpander : public SCEVUseVisitor<SCEVExpander, Value *> {
Value *visitUDivExpr(SCEVUseT<const SCEVUDivExpr *> S);
+ Value *visitSDivExpr(SCEVUseT<const SCEVSDivExpr *> S);
+
Value *visitAddRecExpr(SCEVUseT<const SCEVAddRecExpr *> S);
Value *visitSMaxExpr(SCEVUseT<const SCEVSMaxExpr *> S);
diff --git a/llvm/lib/Analysis/ScalarEvolution.cpp b/llvm/lib/Analysis/ScalarEvolution.cpp
index 6b0951491a88a..53be5fee22fcf 100644
--- a/llvm/lib/Analysis/ScalarEvolution.cpp
+++ b/llvm/lib/Analysis/ScalarEvolution.cpp
@@ -393,6 +393,11 @@ void SCEV::print(raw_ostream &OS) const {
OS << "(" << UDiv->getLHS() << " /u " << UDiv->getRHS() << ")";
return;
}
+ case scSDivExpr: {
+ const SCEVSDivExpr *SDiv = cast<SCEVSDivExpr>(this);
+ OS << "(" << *SDiv->getLHS() << " /s " << *SDiv->getRHS() << ")";
+ return;
+ }
case scUnknown:
cast<SCEVUnknown>(this)->getValue()->printAsOperand(OS, false);
return;
@@ -425,6 +430,8 @@ ArrayRef<SCEVUse> SCEV::operands() const {
return cast<SCEVNAryExpr>(this)->operands();
case scUDivExpr:
return cast<SCEVUDivExpr>(this)->operands();
+ case scSDivExpr:
+ return cast<SCEVSDivExpr>(this)->operands();
case scCouldNotCompute:
llvm_unreachable("Attempt to use a SCEVCouldNotCompute object!");
}
@@ -729,6 +736,7 @@ CompareSCEVComplexity(const LoopInfo *const LI, const SCEV *LHS,
case scAddExpr:
case scMulExpr:
case scUDivExpr:
+ case scSDivExpr:
case scSMaxExpr:
case scUMaxExpr:
case scSMinExpr:
@@ -3044,15 +3052,19 @@ const SCEV *ScalarEvolution::getOrCreateMulExpr(ArrayRef<SCEVUse> Ops,
return S;
}
-const SCEV *ScalarEvolution::getOrCreateUDivExpr(SCEVUse LHS, SCEVUse RHS) {
+const SCEV *ScalarEvolution::getOrCreateDivExpr(bool IsSigned, SCEVUse LHS,
+ SCEVUse RHS) {
FoldingSetNodeID ID;
- ID.AddInteger(scUDivExpr);
+ ID.AddInteger(IsSigned ? scSDivExpr : scUDivExpr);
ID.AddPointer(LHS.getOpaqueValue());
ID.AddPointer(RHS.getOpaqueValue());
void *IP = nullptr;
SCEV *S = UniqueSCEVs.FindNodeOrInsertPos(ID, IP);
if (!S) {
- S = new (SCEVAllocator) SCEVUDivExpr(ID.Intern(SCEVAllocator), LHS, RHS);
+ if (IsSigned)
+ S = new (SCEVAllocator) SCEVSDivExpr(ID.Intern(SCEVAllocator), LHS, RHS);
+ else
+ S = new (SCEVAllocator) SCEVUDivExpr(ID.Intern(SCEVAllocator), LHS, RHS);
UniqueSCEVs.InsertNode(S, IP);
S->computeAndSetCanonical(*this);
registerUser(S, ArrayRef<SCEVUse>({LHS, RHS}));
@@ -3447,27 +3459,42 @@ const SCEV *ScalarEvolution::getURemExpr(SCEVUse LHS, SCEVUse RHS) {
/// Get a canonical unsigned division expression, or something simpler if
/// possible.
-const SCEV *ScalarEvolution::getUDivExpr(SCEVUse LHS, SCEVUse RHS) {
+const SCEV *ScalarEvolution::getDivExpr(bool IsSigned, SCEVUse LHS,
+ SCEVUse RHS) {
assert(!LHS->getType()->isPointerTy() &&
- "SCEVUDivExpr operand can't be pointer!");
+ "SCEVDivExpr operand can't be pointer!");
assert(LHS->getType() == RHS->getType() &&
- "SCEVUDivExpr operand types don't match!");
+ "SCEVDivExpr operand types don't match!");
- if (SCEV *S =
- findExistingSCEVInCache(scUDivExpr, ArrayRef<SCEVUse>({LHS, RHS})))
+ if (SCEV *S = findExistingSCEVInCache(IsSigned ? scSDivExpr : scUDivExpr,
+ ArrayRef<SCEVUse>({LHS, RHS})))
return S;
- // 0 udiv Y == 0
+ if (IsSigned && isKnownNonNegative(LHS) && isKnownNonNegative(RHS))
+ return getDivExpr(false, LHS, RHS);
+
+ // 0/y --> 0
if (match(LHS, m_scev_Zero()))
return LHS;
if (const SCEVConstant *RHSC = dyn_cast<SCEVConstant>(RHS)) {
if (RHSC->getValue()->isOne())
- return LHS; // X udiv 1 --> x
- // If the denominator is zero, the result of the udiv is undefined. Don't
- // try to analyze it, because the resolution chosen here may differ from
- // the resolution chosen in other parts of the compiler.
- if (!RHSC->getValue()->isZero()) {
+ return LHS; // x/1 --> x
+ // If the denominator is zero, the result of the both udiv and sdiv are
+ // undefined. Similarly if the denominator of is -1, and the numerator is
+ // INT_MIN, sdiv is undefined. Don't try to analyze it, because the
+ // resolution chosen here may differ from the resolution chosen in other
+ // parts of the compiler.
+ Type *Ty = LHS->getType();
+ unsigned BW = Ty->getIntegerBitWidth();
+ bool DivIsUndefined = RHSC->getValue()->isZero();
+ bool SDivIsUndefined =
+ RHSC->getValue()->isAllOnesValue() &&
+ getSignedRangeMin(LHS) == APInt::getSignedMinValue(BW);
+ if (!DivIsUndefined && (!IsSigned || !SDivIsUndefined)) {
+ if (IsSigned && RHSC->getValue()->isAllOnesValue())
+ return getNegativeSCEV(LHS); // x/-1 --> -x
+
// Determine if the division can be folded into the operands of
// its operands.
// TODO: Generalize this to non-constants by using known-bits information.
@@ -3479,18 +3506,20 @@ const SCEV *ScalarEvolution::getUDivExpr(SCEVUse LHS, SCEVUse RHS) {
if (!RHSC->getAPInt().isPowerOf2())
++MaxShiftAmt;
IntegerType *ExtTy =
- IntegerType::get(getContext(), getTypeSizeInBits(Ty) + MaxShiftAmt);
- if (const SCEVAddRecExpr *AR = dyn_cast<SCEVAddRecExpr>(LHS))
+ IntegerType::get(getContext(), getTypeSizeInBits(Ty) + MaxShiftAmt);
+ /// TODO: Extend to signed case once we have SRem expressions.
+ if (const SCEVAddRecExpr *AR = dyn_cast<SCEVAddRecExpr>(LHS);
+ AR && !IsSigned)
if (const SCEVConstant *Step =
- dyn_cast<SCEVConstant>(AR->getStepRecurrence(*this))) {
+ dyn_cast<SCEVConstant>(AR->getStepRecurrence(*this))) {
// {X,+,N}/C --> {X/C,+,N/C} if safe and N/C can be folded.
const APInt &StepInt = Step->getAPInt();
const APInt &DivInt = RHSC->getAPInt();
if (!StepInt.urem(DivInt) &&
getZeroExtendExpr(AR, ExtTy) ==
- getAddRecExpr(getZeroExtendExpr(AR->getStart(), ExtTy),
- getZeroExtendExpr(Step, ExtTy),
- AR->getLoop(), SCEV::FlagAnyWrap)) {
+ getAddRecExpr(getZeroExtendExpr(AR->getStart(), ExtTy),
+ getZeroExtendExpr(Step, ExtTy), AR->getLoop(),
+ SCEV::FlagAnyWrap)) {
SmallVector<SCEVUse, 4> Operands;
for (const SCEV *Op : AR->operands())
Operands.push_back(getUDivExpr(Op, RHS));
@@ -3529,12 +3558,12 @@ const SCEV *ScalarEvolution::getUDivExpr(SCEVUse LHS, SCEVUse RHS) {
}
// (A*B)/C --> A*(B/C) if safe and B/C can be folded.
if (const SCEVMulExpr *M = dyn_cast<SCEVMulExpr>(LHS)) {
- if (M->hasNoUnsignedWrap()) {
+ if (IsSigned ? M->hasNoSignedWrap() : M->hasNoUnsignedWrap()) {
// Find an operand that's safely divisible.
for (unsigned i = 0, e = M->getNumOperands(); i != e; ++i) {
const SCEV *Op = M->getOperand(i);
- const SCEV *Div = getUDivExpr(Op, RHSC);
- if (!isa<SCEVUDivExpr>(Div) && getMulExpr(Div, RHSC) == Op) {
+ const SCEV *Div = getDivExpr(IsSigned, Op, RHSC);
+ if (!isa<SCEVDivExpr>(Div) && getMulExpr(Div, RHSC) == Op) {
SmallVector<SCEVUse, 4> Operands(M->operands());
Operands[i] = Div;
return getMulExpr(Operands);
@@ -3543,43 +3572,47 @@ const SCEV *ScalarEvolution::getUDivExpr(SCEVUse LHS, SCEVUse RHS) {
// Even if it's not divisible, try to remove a common factor.
if (const auto *LHSC = dyn_cast<SCEVConstant>(M->getOperand(0))) {
- APInt Factor = APIntOps::GreatestCommonDivisor(LHSC->getAPInt(),
- RHSC->getAPInt());
+ APInt Factor = APIntOps::GreatestCommonDivisor(
+ LHSC->getAPInt(), RHSC->getAPInt(), IsSigned);
if (!Factor.isIntN(1)) {
SmallVector<SCEVUse, 2> NewOperands;
- NewOperands.push_back(getConstant(LHSC->getAPInt().udiv(Factor)));
+ NewOperands.push_back(
+ IsSigned ? getConstant(LHSC->getAPInt().sdiv(Factor))
+ : getConstant(LHSC->getAPInt().udiv(Factor)));
append_range(NewOperands, M->operands().drop_front());
const SCEV *NewMul = getMulExpr(NewOperands);
- return getUDivExpr(NewMul,
- getConstant(RHSC->getAPInt().udiv(Factor)));
+ return getDivExpr(
+ IsSigned, NewMul,
+ IsSigned ? getConstant(RHSC->getAPInt().sdiv(Factor))
+ : getConstant(RHSC->getAPInt().udiv(Factor)));
}
}
}
}
// (A/B)/C --> A/(B*C) if safe and B*C can be folded.
- if (const SCEVUDivExpr *OtherDiv = dyn_cast<SCEVUDivExpr>(LHS)) {
+ if (auto *OtherDiv = dyn_cast<SCEVDivExpr>(LHS)) {
if (auto *DivisorConstant =
dyn_cast<SCEVConstant>(OtherDiv->getRHS())) {
bool Overflow = false;
- APInt NewRHS =
- DivisorConstant->getAPInt().umul_ov(RHSC->getAPInt(), Overflow);
- if (Overflow) {
+ APInt NewRHS = IsSigned ? DivisorConstant->getAPInt().smul_ov(
+ RHSC->getAPInt(), Overflow)
+ : DivisorConstant->getAPInt().umul_ov(
+ RHSC->getAPInt(), Overflow);
+ if (Overflow)
return getConstant(RHSC->getType(), 0, false);
- }
- return getUDivExpr(OtherDiv->getLHS(), getConstant(NewRHS));
+ return getDivExpr(IsSigned, OtherDiv->getLHS(), getConstant(NewRHS));
}
}
// (A+B)/C --> (A/C + B/C) if the add does not unsigned wrap and A/C and
// B/C can be folded.
if (const SCEVAddExpr *A = dyn_cast<SCEVAddExpr>(LHS)) {
- if (A->hasNoUnsignedWrap()) {
+ if (IsSigned ? A->hasNoSignedWrap() : A->hasNoUnsignedWrap()) {
SmallVector<SCEVUse, 4> Operands;
for (unsigned i = 0, e = A->getNumOperands(); i != e; ++i) {
- const SCEV *Op = getUDivExpr(A->getOperand(i), RHS);
- if (isa<SCEVUDivExpr>(Op) ||
- getMulExpr(Op, RHS) != A->getOperand(i))
+ const SCEV *Op = getDivExpr(IsSigned, A->getOperand(i), RHS);
+ if (isa<SCEVDivExpr>(Op) || getMulExpr(Op, RHS) != A->getOperand(i))
break;
Operands.push_back(Op);
}
@@ -3600,9 +3633,11 @@ const SCEV *ScalarEvolution::getUDivExpr(SCEVUse LHS, SCEVUse RHS) {
const SCEV *A;
if (match(LHS, m_scev_Add(m_scev_APInt(NMinusM),
m_scev_Mul(m_scev_APInt(M), m_SCEV(A))))) {
- if (N.isPowerOf2() && M->isPowerOf2() && M->ult(N) &&
- *NMinusM == N - *M) {
- return getUDivExpr(
+ if ((IsSigned ? N.isNegatedPowerOf2() : N.isPowerOf2()) &&
+ (IsSigned ? M->isNegatedPowerOf2() : M->isPowerOf2()) &&
+ M->ult(N) && *NMinusM == N - *M) {
+ return getDivExpr(
+ IsSigned,
getAddExpr(getConstant(N - 1), getMulExpr(getConstant(*M), A)),
RHS);
}
@@ -3610,7 +3645,8 @@ const SCEV *ScalarEvolution::getUDivExpr(SCEVUse LHS, SCEVUse RHS) {
// Fold if both operands are constant.
if (const SCEVConstant *LHSC = dyn_cast<SCEVConstant>(LHS))
- return getConstant(LHSC->getAPInt().udiv(RHSC->getAPInt()));
+ return getConstant(IsSigned ? LHSC->getAPInt().sdiv(RHSC->getAPInt())
+ : LHSC->getAPInt().udiv(RHSC->getAPInt()));
}
}
@@ -3622,9 +3658,9 @@ const SCEV *ScalarEvolution::getUDivExpr(SCEVUse LHS, SCEVUse RHS) {
NegC->isNegative() && !NegC->isMinSignedValue() && *C == -*NegC)
return getZero(LHS->getType());
- // (%a * %b)<nuw> / %b -> %a
+ // (%a * %b)<nuw or nsw> / %b -> %a
const auto *Mul = dyn_cast<SCEVMulExpr>(LHS);
- if (Mul && Mul->hasNoUnsignedWrap()) {
+ if (Mul && (IsSigned ? Mul->hasNoSignedWrap() : Mul->hasNoUnsignedWrap())) {
for (int i = 0, e = Mul->getNumOperands(); i != e; ++i) {
if (Mul->getOperand(i) == RHS) {
SmallVector<SCEVUse, 2> Operands;
@@ -3636,13 +3672,24 @@ const SCEV *ScalarEvolution::getUDivExpr(SCEVUse LHS, SCEVUse RHS) {
}
// TODO: Generalize to handle any common factors.
- // udiv (mul nuw a, vscale), (mul nuw b, vscale) --> udiv a, b
+ // (mul nuw a, vscale)/(mul nuw b, vscale) --> a/b
const SCEV *NewLHS, *NewRHS;
- if (match(LHS, m_scev_c_NUWMul(m_SCEV(NewLHS), m_SCEVVScale())) &&
- match(RHS, m_scev_c_NUWMul(m_SCEV(NewRHS), m_SCEVVScale())))
- return getUDivExpr(NewLHS, NewRHS);
+ if (IsSigned
+ ? match(LHS, m_scev_c_NSWMul(m_SCEV(NewLHS), m_SCEVVScale())) &&
+ match(RHS, m_scev_c_NSWMul(m_SCEV(NewRHS), m_SCEVVScale()))
+ : match(LHS, m_scev_c_NUWMul(m_SCEV(NewLHS), m_SCEVVScale())) &&
+ match(RHS, m_scev_c_NUWMul(m_SCEV(NewRHS), m_SCEVVScale())))
+ return getDivExpr(IsSigned, NewLHS, NewRHS);
+
+ return getOrCreateDivExpr(IsSigned, LHS, RHS);
+}
- return getOrCreateUDivExpr(LHS, RHS);
+const SCEV *ScalarEvolution::getUDivExpr(SCEVUse LHS, SCEVUse RHS) {
+ return getDivExpr(/*IsSigned=*/false, LHS, RHS);
+}
+
+const SCEV *ScalarEvolution::getSDivExpr(SCEVUse LHS, SCEVUse RHS) {
+ return getDivExpr(/*IsSigned=*/true, LHS, RHS);
}
/// Get a canonical unsigned division expression, or something simpler if
@@ -4069,6 +4116,8 @@ class SCEVSequentialMinMaxDeduplicatingVisitor final
RetVal visitUDivExpr(const SCEVUDivExpr *Expr) { return Expr; }
+ RetVal visitSDivExpr(const SCEVSDivExpr *Expr) { return Expr; }
+
RetVal visitAddRecExpr(const SCEVAddRecExpr *Expr) { return Expr; }
RetVal visitSMaxExpr(const SCEVSMaxExpr *Expr) {
@@ -4109,6 +4158,7 @@ static bool scevUnconditionallyPropagatesPoisonFromOperands(SCEVTypes Kind) {
case scAddExpr:
case scMulExpr:
case scUDivExpr:
+ case scSDivExpr:
case scAddRecExpr:
case scUMaxExpr:
case scSMaxExpr:
@@ -6319,6 +6369,7 @@ APInt ScalarEvolution::getConstantMultipleImpl(const SCEV *S,
case scPtrToAddr:
return getConstantMultiple(cast<SCEVCastExpr>(S)->getOperand());
case scUDivExpr:
+ case scSDivExpr:
case scVScale:
return APInt(BitWidth, 1);
case scTruncate: {
@@ -6624,6 +6675,7 @@ ScalarEvolution::getRangeRefIter(const SCEV *S,
case scAddExpr:
case scMulExpr:
case scUDivExpr:
+ case scSDivExpr:
case scAddRecExpr:
case scUMaxExpr:
case scSMaxExpr:
@@ -6792,6 +6844,13 @@ const ConstantRange &ScalarEvolution::getRangeRef(
return setRange(UDiv, SignHint,
ConservativeResult.intersectWith(X.udiv(Y), RangeType));
}
+ case scSDivExpr: {
+ const SCEVSDivExpr *SDiv = cast<SCEVSDivExpr>(S);
+ ConstantRange X = getRangeRef(SDiv->getLHS(), SignHint, Depth + 1);
+ ConstantRange Y = getRangeRef(SDiv->getRHS(), SignHint, Depth + 1);
+ return setRange(SDiv, SignHint,
+ ConservativeResult.intersectWith(X.sdiv(Y), RangeType));
+ }
case scAddRecExpr: {
const SCEVAddRecExpr *AddRec = cast<SCEVAddRecExpr>(S);
// If there's no unsigned wrap, the value will never be less than its
@@ -8264,11 +8323,7 @@ const SCEV *ScalarEvolution::createSCEV(Value *V) {
return getUnknown(V);
case Instruction::SDiv:
- // If both operands are non-negative, this is just an udiv.
- if (isKnownNonNegative(getSCEV(U->getOperand(0))) &&
- isKnownNonNegative(getSCEV(U->getOperand(1))))
- return getUDivExpr(getSCEV(U->getOperand(0)), getSCEV(U->getOperand(1)));
- break;
+ return getSDivExpr(getSCEV(U->getOperand(0)), getSCEV(U->getOperand(1)));
case Instruction::SRem:
// If both operands are non-negative, this is just an urem.
@@ -10123,6 +10178,7 @@ static Constant *BuildConstantFromSCEV(const SCEV *V) {
case scSignExtend:
case scZeroExtend:
case scUDivExpr:
+ case scSDivExpr:
case scSMaxExpr:
case scUMaxExpr:
case scSMinExpr:
@@ -10151,6 +10207,8 @@ const SCEV *ScalarEvolution::getWithOperands(const SCEV *S,
return getMulExpr(NewOps, cast<SCEVMulExpr>(S)->getNoWrapFlags());
case scUDivExpr:
return getUDivExpr(NewOps[0], NewOps[1]);
+ case scSDivExpr:
+ return getSDivExpr(NewOps[0], NewOps[1]);
case scUMaxExpr:
case scSMaxExpr:
case scUMinExpr:
@@ -10227,6 +10285,7 @@ const SCEV *ScalarEvolution::computeSCEVAtScope(const SCEV *V, const Loop *L) {
case scAddExpr:
case scMulExpr:
case scUDivExpr:
+ case scSDivExpr:
case scUMaxExpr:
case scSMaxExpr:
case scUMinExpr:
@@ -13010,74 +13069,67 @@ bool ScalarEvolution::isImpliedViaOperations(CmpPredicate Pred, const SCEV *LHS,
// (LHS = LL + LR) && (LR >= 0) && (LL > RHS) => (LHS > RHS).
if (IsSumGreaterThanRHS(LL, LR) || IsSumGreaterThanRHS(LR, LL))
return true;
- } else if (auto *LHSUnknownExpr = dyn_cast<SCEVUnknown>(LHS)) {
- Value *LL, *LR;
- // FIXME: Once we have SDiv implemented, we can get rid of this matching.
-
- using namespace llvm::PatternMatch;
-
- if (match(LHSUnknownExpr->getValue(), m_SDiv(m_Value(LL), m_Value(LR)))) {
- // Rules for division.
- // We are going to perform some comparisons with Denominator and its
- // derivative expressions. In general case, creating a SCEV for it may
- // lead to a complex analysis of the entire graph, and in particular it
- // can request trip count recalculation for the same loop. This would
- // cache as SCEVCouldNotCompute to avoid the infinite recursion. To avoid
- // this, we only want to create SCEVs that are constants in this section.
- // So we bail if Denominator is not a constant.
- if (!isa<ConstantInt>(LR))
- return false;
+ } else if (auto *LHSSDivExpr = dyn_cast<SCEVSDivExpr>(LHS)) {
+ SCEVUse LL = LHSSDivExpr->getOperand(0);
+ SCEVUse LR = LHSSDivExpr->getOperand(0);
+
+ // Rules for division.
+ // We are going to perform some comparisons with Denominator and its
+ // derivative expressions. In general case, creating a SCEV for it may
+ // lead to a complex analysis of the entire graph, and in particular it
+ // can request trip count recalculation for the same loop. This would
+ // cache as SCEVCouldNotCompute to avoid the infinite recursion. To avoid
+ // this, we only want to create SCEVs that are constants in this section.
+ // So we bail if Denominator is not a constant.
+ auto *Denominator = dyn_cast<SCEVConstant>(LR);
+ if (!Denominator)
+ return false;
- auto *Denominator = cast<SCEVConstant>(getSCEV(LR));
+ // We want to make sure that LHS = FoundLHS / Denominator. If it is so,
+ // then a SCEV for the numerator already exists and matches with FoundLHS.
+ SCEVUse Numerator = LL;
+ if (Numerator->getType() != FoundLHS->getType())
+ return false;
- // We want to make sure that LHS = FoundLHS / Denominator. If it is so,
- // then a SCEV for the numerator already exists and matches with FoundLHS.
- auto *Numerator = getExistingSCEV(LL);
- if (!Numerator || Numerator->getType() != FoundLHS->getType())
- return false;
+ // Make sure that the numerator matches with FoundLHS and the denominator
+ // is positive.
+ if (!HasSameValue(Numerator, FoundLHS) || !isKnownPositive(Denominator))
+ return false;
- // Make sure that the numerator matches with FoundLHS and the denominator
- // is positive.
- if (!HasSameValue(Numerator, FoundLHS) || !isKnownPositive(Denominator))
- return false;
+ auto *DTy = Denominator->getType();
+ auto *FRHSTy = FoundRHS->getType();
+ if (DTy->isPointerTy() != FRHSTy->isPointerTy())
+ // One of types is a pointer and another one is not. We cannot extend
+ // them properly to a wider type, so let us just reject this case.
+ // TODO: Usage of getEffectiveSCEVType for DTy, FRHSTy etc should help
+ // to avoid this check.
+ return false;
- auto *DTy = Denominator->getType();
- auto *FRHSTy = FoundRHS->getType();
- if (DTy->isPointerTy() != FRHSTy->isPointerTy())
- // One of types is a pointer and another one is not. We cannot extend
- // them properly to a wider type, so let us just reject this case.
- // TODO: Usage of getEffectiveSCEVType for DTy, FRHSTy etc should help
- // to avoid this check.
- return false;
+ // Given that:
+ // FoundLHS > FoundRHS, LHS = FoundLHS / Denominator, Denominator > 0.
+ auto *WTy = getWiderType(DTy, FRHSTy);
+ auto *DenominatorExt = getNoopOrSignExtend(Denominator, WTy);
+ auto *FoundRHSExt = getNoopOrSignExtend(FoundRHS, WTy);
- // Given that:
- // FoundLHS > FoundRHS, LHS = FoundLHS / Denominator, Denominator > 0.
- auto *WTy = getWiderType(DTy, FRHSTy);
- auto *DenominatorExt = getNoopOrSignExtend(Denominator, WTy);
- auto *FoundRHSExt = getNoopOrSignExtend(FoundRHS, WTy);
-
- // Try to prove the following rule:
- // (FoundRHS > Denominator - 2) && (RHS <= 0) => (LHS > RHS).
- // For example, given that FoundLHS > 2. It means that FoundLHS is at
- // least 3. If we divide it by Denominator < 4, we will have at least 1.
- auto *DenomMinusTwo = getMinusSCEV(DenominatorExt, getConstant(WTy, 2));
- if (isKnownNonPositive(RHS) &&
- IsSGTViaContext(FoundRHSExt, DenomMinusTwo))
- return true;
+ // Try to prove the following rule:
+ // (FoundRHS > Denominator - 2) && (RHS <= 0) => (LHS > RHS).
+ // For example, given that FoundLHS > 2. It means that FoundLHS is at
+ // least 3. If we divide it by Denominator < 4, we will have at least 1.
+ auto *DenomMinusTwo = getMinusSCEV(DenominatorExt, getConstant(WTy, 2));
+ if (isKnownNonPositive(RHS) && IsSGTViaContext(FoundRHSExt, DenomMinusTwo))
+ return true;
- // Try to prove the following rule:
- // (FoundRHS > -1 - Denominator) && (RHS < 0) => (LHS > RHS).
- // For example, given that FoundLHS > -3. Then FoundLHS is at least -2.
- // If we divide it by Denominator > 2, then:
- // 1. If FoundLHS is negative, then the result is 0.
- // 2. If FoundLHS is non-negative, then the result is non-negative.
- // Anyways, the result is non-negative.
- auto *MinusOne = getMinusOne(WTy);
- auto *NegDenomMinusOne = getMinusSCEV(MinusOne, DenominatorExt);
- if (isKnownNegative(RHS) &&
- IsSGTViaContext(FoundRHSExt, NegDenomMinusOne))
- return true;
- }
+ // Try to prove the following rule:
+ // (FoundRHS > -1 - Denominator) && (RHS < 0) => (LHS > RHS).
+ // For example, given that FoundLHS > -3. Then FoundLHS is at least -2.
+ // If we divide it by Denominator > 2, then:
+ // 1. If FoundLHS is negative, then the result is 0.
+ // 2. If FoundLHS is non-negative, then the result is non-negative.
+ // Anyways, the result is non-negative.
+ auto *MinusOne = getMinusOne(WTy);
+ auto *NegDenomMinusOne = getMinusSCEV(MinusOne, DenominatorExt);
+ if (isKnownNegative(RHS) && IsSGTViaContext(FoundRHSExt, NegDenomMinusOne))
+ return true;
}
// If our expression contained SCEVUnknown Phis, and we split it down and now
@@ -14459,6 +14511,7 @@ ScalarEvolution::computeLoopDisposition(const SCEV *S, const Loop *L) {
case scAddExpr:
case scMulExpr:
case scUDivExpr:
+ case scSDivExpr:
case scUMaxExpr:
case scSMaxExpr:
case scUMinExpr:
@@ -14549,6 +14602,7 @@ ScalarEvolution::computeBlockDisposition(const SCEV *S, const BasicBlock *BB) {
case scAddExpr:
case scMulExpr:
case scUDivExpr:
+ case scSDivExpr:
case scUMaxExpr:
case scSMaxExpr:
case scUMinExpr:
diff --git a/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp b/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
index f01a674609325..b3c8676066ac3 100644
--- a/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
+++ b/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
@@ -463,6 +463,7 @@ const Loop *SCEVExpander::getRelevantLoop(const SCEV *S) {
case scAddExpr:
case scMulExpr:
case scUDivExpr:
+ case scSDivExpr:
case scAddRecExpr:
case scUMaxExpr:
case scSMaxExpr:
@@ -729,20 +730,29 @@ Value *SCEVExpander::visitUDivExpr(SCEVUseT<const SCEVUDivExpr *> S) {
const SCEV *RHSExpr = S->getRHS();
Value *RHS = expand(RHSExpr);
if (SafeUDivMode) {
- bool GuaranteedNotPoison =
- ScalarEvolution::isGuaranteedNotToBePoison(RHSExpr);
- if (!GuaranteedNotPoison)
+ if (!ScalarEvolution::isGuaranteedNotToBePoison(RHSExpr))
RHS = Builder.CreateFreeze(RHS);
// We need an umax if either RHSExpr is not known to be zero, or if it is
// not guaranteed to be non-poison. In the later case, the frozen poison may
// be 0.
- if (!SE.isKnownNonZero(RHSExpr) || !GuaranteedNotPoison)
+ if (S->mayTriggerUB(SE))
RHS = Builder.CreateIntrinsic(RHS->getType(), Intrinsic::umax,
{RHS, ConstantInt::get(RHS->getType(), 1)});
}
return InsertBinop(Instruction::UDiv, LHS, RHS, SCEV::FlagAnyWrap,
- /*IsSafeToHoist*/ SE.isKnownNonZero(S->getRHS()));
+ /*IsSafeToHoist=*/!S->mayTriggerUB(SE));
+}
+
+Value *SCEVExpander::visitSDivExpr(SCEVUseT<const SCEVSDivExpr *> S) {
+ Value *LHS = expand(S->getLHS());
+ Value *RHS = expand(S->getRHS());
+ if (S->mayTriggerUB(SE)) {
+ LHS = Builder.CreateFreeze(LHS);
+ RHS = Builder.CreateFreeze(RHS);
+ }
+ return InsertBinop(Instruction::SDiv, LHS, RHS, SCEV::FlagAnyWrap,
+ /*IsSafeToHoist=*/!S->mayTriggerUB(SE));
}
/// Determine if this is a well-behaved chain of instructions leading back to
@@ -1693,20 +1703,12 @@ Value *SCEVExpander::expand(SCEVUse S) {
// We can move insertion point only if there is no div or rem operations
// otherwise we are risky to move it over the check for zero denominator.
- auto SafeToHoist = [](const SCEV *S) {
- return !SCEVExprContains(S, [](const SCEV *S) {
- if (const auto *D = dyn_cast<SCEVUDivExpr>(S)) {
- if (const auto *SC = dyn_cast<SCEVConstant>(D->getRHS()))
- // Division by non-zero constants can be hoisted.
- return SC->getValue()->isZero();
- // All other divisions should not be moved as they may be
- // divisions by zero and should be kept within the
- // conditions of the surrounding loops that guard their
- // execution (see PR35406).
- return true;
- }
- return false;
- });
+ auto SafeToHoist = [this](const SCEV *S) {
+ return !SCEVExprContains(S, [this](const SCEV *S) {
+ const auto *D = dyn_cast<SCEVDivExpr>(S);
+ // TODO: Why is this RHS-constant check necessary?
+ return D && (!isa<SCEVConstant>(D->getRHS()) || D->mayTriggerUB(SE));
+ });
};
if (SafeToHoist(S)) {
for (Loop *L = SE.LI.getLoopFor(Builder.GetInsertBlock());;
@@ -2106,6 +2108,11 @@ template<typename T> static InstructionCost costAndCollectOperands(
Cost = ArithCost(Opcode, 1);
break;
}
+ case scSDivExpr: {
+ unsigned Opcode = Instruction::SDiv;
+ Cost = ArithCost(Opcode, 1);
+ break;
+ }
case scAddExpr:
Cost = ArithCost(Instruction::Add, S->getNumOperands() - 1);
break;
@@ -2230,7 +2237,8 @@ bool SCEVExpander::isHighCostExpansionHelper(
costAndCollectOperands<SCEVCastExpr>(WorkItem, TTI, CostKind, Worklist);
return false; // Will answer upon next entry into this function.
}
- case scUDivExpr: {
+ case scUDivExpr:
+ case scSDivExpr: {
// UDivExpr is very likely a UDiv that ScalarEvolution's HowFarToZero or
// HowManyLessThans produced to compute a precise expression, rather than a
// UDiv from the user's code. If we can't find a UDiv in the code with some
@@ -2244,7 +2252,7 @@ bool SCEVExpander::isHighCostExpansionHelper(
return false; // Consider it to be free.
Cost +=
- costAndCollectOperands<SCEVUDivExpr>(WorkItem, TTI, CostKind, Worklist);
+ costAndCollectOperands<SCEVDivExpr>(WorkItem, TTI, CostKind, Worklist);
return false; // Will answer upon next entry into this function.
}
case scAddExpr:
@@ -2535,9 +2543,8 @@ struct SCEVFindUnsafe {
: SE(SE), CanonicalMode(CanonicalMode) {}
bool follow(const SCEV *S) {
- if (const SCEVUDivExpr *D = dyn_cast<SCEVUDivExpr>(S)) {
- if (!SE.isKnownNonZero(D->getRHS()) ||
- !SE.isGuaranteedNotToBePoison(D->getRHS())) {
+ if (auto *D = dyn_cast<SCEVDivExpr>(S)) {
+ if (D->mayTriggerUB(SE)) {
IsUnsafe = true;
return false;
}
diff --git a/llvm/test/Analysis/ScalarEvolution/add-expr-pointer-operand-sorting.ll b/llvm/test/Analysis/ScalarEvolution/add-expr-pointer-operand-sorting.ll
index 5dffee19cb312..90c59720986ce 100644
--- a/llvm/test/Analysis/ScalarEvolution/add-expr-pointer-operand-sorting.ll
+++ b/llvm/test/Analysis/ScalarEvolution/add-expr-pointer-operand-sorting.ll
@@ -34,9 +34,9 @@ define i32 @d(i32 %base) {
; CHECK-NEXT: %sub.ptr.sub = sub i64 %sub.ptr.lhs.cast, ptrtoint (ptr @b to i64)
; CHECK-NEXT: --> ((-1 * (ptrtoaddr ptr @b to i64)) + (ptrtoaddr ptr %load1 to i64)) U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %for.cond: Variant }
; CHECK-NEXT: %sub.ptr.div = sdiv exact i64 %sub.ptr.sub, 4
-; CHECK-NEXT: --> %sub.ptr.div U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %for.cond: Variant }
+; CHECK-NEXT: --> (((-1 * (ptrtoaddr ptr @b to i64)) + (ptrtoaddr ptr %load1 to i64)) /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %for.cond: Variant }
; CHECK-NEXT: %arrayidx1 = getelementptr inbounds [1 x i8], ptr %arrayidx, i64 0, i64 %sub.ptr.div
-; CHECK-NEXT: --> ({((sext i32 %base to i64) + %e),+,1}<nw><%for.cond> + %sub.ptr.div) U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %for.cond: Variant }
+; CHECK-NEXT: --> ((((-1 * (ptrtoaddr ptr @b to i64)) + (ptrtoaddr ptr %load1 to i64)) /s 4) + {((sext i32 %base to i64) + %e),+,1}<nw><%for.cond>) U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %for.cond: Variant }
; CHECK-NEXT: %load2 = load i8, ptr %arrayidx1, align 1
; CHECK-NEXT: --> %load2 U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %for.cond: Variant }
; CHECK-NEXT: %conv = sext i8 %load2 to i32
diff --git a/llvm/test/Analysis/ScalarEvolution/addrec-may-wrap-sdiv-canonicalize.ll b/llvm/test/Analysis/ScalarEvolution/addrec-may-wrap-sdiv-canonicalize.ll
new file mode 100644
index 0000000000000..d0cafc14ee099
--- /dev/null
+++ b/llvm/test/Analysis/ScalarEvolution/addrec-may-wrap-sdiv-canonicalize.ll
@@ -0,0 +1,402 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes='print<scalar-evolution>' -disable-output 2>&1 | FileCheck %s
+
+declare void @use(i64)
+
+define void @test_step2_div4(i64 %n) {
+; CHECK-LABEL: 'test_step2_div4'
+; CHECK-NEXT: Classifying expressions for: @test_step2_div4
+; CHECK-NEXT: %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+; CHECK-NEXT: --> {0,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.0 = sdiv i64 %iv, 4
+; CHECK-NEXT: --> ({0,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.1 = add i64 %iv, 1
+; CHECK-NEXT: --> {1,+,2}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.1 = sdiv i64 %iv.1, 4
+; CHECK-NEXT: --> ({1,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.2 = add i64 %iv, 2
+; CHECK-NEXT: --> {2,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.2 = sdiv i64 %iv.2, 4
+; CHECK-NEXT: --> ({2,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.neg.1 = add i64 %iv, -1
+; CHECK-NEXT: --> {-1,+,2}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.neg.1 = sdiv i64 %iv.neg.1, 4
+; CHECK-NEXT: --> ({-1,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.next = add i64 %iv, 2
+; CHECK-NEXT: --> {2,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: Determining loop execution counts for: @test_step2_div4
+; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %div.0 = sdiv i64 %iv, 4
+ call void @use(i64 %div.0)
+ %iv.1 = add i64 %iv, 1
+ %div.1 = sdiv i64 %iv.1, 4
+ call void @use(i64 %div.1)
+ %iv.2 = add i64 %iv, 2
+ %div.2 = sdiv i64 %iv.2, 4
+ call void @use(i64 %div.2)
+ %iv.neg.1 = add i64 %iv, -1
+ %div.neg.1 = sdiv i64 %iv.neg.1, 4
+ call void @use(i64 %div.neg.1)
+ %iv.next = add i64 %iv, 2
+ %cond = icmp slt i64 %iv, %n
+ br i1 %cond, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+define void @test_step3_div6(i64 %n) {
+; CHECK-LABEL: 'test_step3_div6'
+; CHECK-NEXT: Classifying expressions for: @test_step3_div6
+; CHECK-NEXT: %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+; CHECK-NEXT: --> {0,+,3}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.0 = sdiv i64 %iv, 6
+; CHECK-NEXT: --> ({0,+,3}<%loop> /s 6) U: [-1537228672809129301,1537228672809129302) S: [-1537228672809129301,1537228672809129302) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.1 = add i64 %iv, 1
+; CHECK-NEXT: --> {1,+,3}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.1 = sdiv i64 %iv.1, 6
+; CHECK-NEXT: --> ({1,+,3}<%loop> /s 6) U: [-1537228672809129301,1537228672809129302) S: [-1537228672809129301,1537228672809129302) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.2 = add i64 %iv, 2
+; CHECK-NEXT: --> {2,+,3}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.2 = sdiv i64 %iv.2, 6
+; CHECK-NEXT: --> ({2,+,3}<%loop> /s 6) U: [-1537228672809129301,1537228672809129302) S: [-1537228672809129301,1537228672809129302) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.neg.1 = add i64 %iv, -1
+; CHECK-NEXT: --> {-1,+,3}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.neg.1 = sdiv i64 %iv.neg.1, 6
+; CHECK-NEXT: --> ({-1,+,3}<%loop> /s 6) U: [-1537228672809129301,1537228672809129302) S: [-1537228672809129301,1537228672809129302) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.next = add i64 %iv, 3
+; CHECK-NEXT: --> {3,+,3}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: Determining loop execution counts for: @test_step3_div6
+; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %div.0 = sdiv i64 %iv, 6
+ call void @use(i64 %div.0)
+ %iv.1 = add i64 %iv, 1
+ %div.1 = sdiv i64 %iv.1, 6
+ call void @use(i64 %div.1)
+ %iv.2 = add i64 %iv, 2
+ %div.2 = sdiv i64 %iv.2, 6
+ call void @use(i64 %div.2)
+ %iv.neg.1 = add i64 %iv, -1
+ %div.neg.1 = sdiv i64 %iv.neg.1, 6
+ call void @use(i64 %div.neg.1)
+ %iv.next = add i64 %iv, 3
+ %cond = icmp slt i64 %iv, %n
+ br i1 %cond, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+
+define void @test_step4_div4(i64 %n) {
+; CHECK-LABEL: 'test_step4_div4'
+; CHECK-NEXT: Classifying expressions for: @test_step4_div4
+; CHECK-NEXT: %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+; CHECK-NEXT: --> {0,+,4}<%loop> U: [0,-3) S: [-9223372036854775808,9223372036854775805) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.0 = sdiv i64 %iv, 4
+; CHECK-NEXT: --> ({0,+,4}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.1 = add i64 %iv, 1
+; CHECK-NEXT: --> {1,+,4}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.1 = sdiv i64 %iv.1, 4
+; CHECK-NEXT: --> ({1,+,4}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.2 = add i64 %iv, 2
+; CHECK-NEXT: --> {2,+,4}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.2 = sdiv i64 %iv.2, 4
+; CHECK-NEXT: --> ({2,+,4}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.3 = add i64 %iv, 3
+; CHECK-NEXT: --> {3,+,4}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.3 = sdiv i64 %iv.3, 4
+; CHECK-NEXT: --> ({3,+,4}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.4 = add i64 %iv, 4
+; CHECK-NEXT: --> {4,+,4}<%loop> U: [0,-3) S: [-9223372036854775808,9223372036854775805) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.4 = sdiv i64 %iv.4, 4
+; CHECK-NEXT: --> ({4,+,4}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.5 = add i64 %iv, 5
+; CHECK-NEXT: --> {5,+,4}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %div.5 = sdiv i64 %iv.5, 4
+; CHECK-NEXT: --> ({5,+,4}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.next = add i64 %iv, 4
+; CHECK-NEXT: --> {4,+,4}<%loop> U: [0,-3) S: [-9223372036854775808,9223372036854775805) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: Determining loop execution counts for: @test_step4_div4
+; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %div.0 = sdiv i64 %iv, 4
+ call void @use(i64 %div.0)
+ %iv.1 = add i64 %iv, 1
+ %div.1 = sdiv i64 %iv.1, 4
+ call void @use(i64 %div.1)
+ %iv.2 = add i64 %iv, 2
+ %div.2 = sdiv i64 %iv.2, 4
+ call void @use(i64 %div.2)
+ %iv.3 = add i64 %iv, 3
+ %div.3 = sdiv i64 %iv.3, 4
+ call void @use(i64 %div.3)
+ %iv.4 = add i64 %iv, 4
+ %div.4 = sdiv i64 %iv.4, 4
+ call void @use(i64 %div.4)
+ %iv.5 = add i64 %iv, 5
+ %div.5 = sdiv i64 %iv.5, 4
+ call void @use(i64 %div.5)
+ %iv.next = add i64 %iv, 4
+ %cond = icmp slt i64 %iv, %n
+ br i1 %cond, label %loop, label %exit
+
+exit:
+ ret void
+}
+
+define void @test_step2_start_outer_add_rec_step_16(i64 %n, i64 %m) {
+; CHECK-LABEL: 'test_step2_start_outer_add_rec_step_16'
+; CHECK-NEXT: Classifying expressions for: @test_step2_start_outer_add_rec_step_16
+; CHECK-NEXT: %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+; CHECK-NEXT: --> {0,+,16}<%outer.header> U: [0,-15) S: [-9223372036854775808,9223372036854775793) Exits: <<Unknown>> LoopDispositions: { %outer.header: Computable, %loop: Invariant }
+; CHECK-NEXT: %iv = phi i64 [ %outer.iv, %outer.header ], [ %iv.next, %loop ]
+; CHECK-NEXT: --> {{\{\{}}0,+,16}<%outer.header>,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.0 = sdiv i64 %iv, 4
+; CHECK-NEXT: --> ({{\{\{}}0,+,16}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.1 = add i64 %iv, 1
+; CHECK-NEXT: --> {{\{\{}}1,+,16}<%outer.header>,+,2}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.1 = sdiv i64 %iv.1, 4
+; CHECK-NEXT: --> ({{\{\{}}1,+,16}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.2 = add i64 %iv, 2
+; CHECK-NEXT: --> {{\{\{}}2,+,16}<%outer.header>,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.2 = sdiv i64 %iv.2, 4
+; CHECK-NEXT: --> ({{\{\{}}2,+,16}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.3 = add i64 %iv, 3
+; CHECK-NEXT: --> {{\{\{}}3,+,16}<%outer.header>,+,2}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.3 = sdiv i64 %iv.3, 4
+; CHECK-NEXT: --> ({{\{\{}}3,+,16}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.4 = add i64 %iv, 4
+; CHECK-NEXT: --> {{\{\{}}4,+,16}<%outer.header>,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.4 = sdiv i64 %iv.4, 4
+; CHECK-NEXT: --> ({{\{\{}}4,+,16}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.5 = add i64 %iv, 5
+; CHECK-NEXT: --> {{\{\{}}5,+,16}<%outer.header>,+,2}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.5 = sdiv i64 %iv.5, 4
+; CHECK-NEXT: --> ({{\{\{}}5,+,16}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.neg.1 = add i64 %iv, -1
+; CHECK-NEXT: --> {{\{\{}}-1,+,16}<%outer.header>,+,2}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.neg.1 = sdiv i64 %iv.neg.1, 4
+; CHECK-NEXT: --> ({{\{\{}}-1,+,16}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div3.0 = sdiv i64 %iv, 3
+; CHECK-NEXT: --> ({{\{\{}}0,+,16}<%outer.header>,+,2}<%loop> /s 3) U: [-3074457345618258602,3074457345618258603) S: [-3074457345618258602,3074457345618258603) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div3.1 = sdiv i64 %iv.1, 3
+; CHECK-NEXT: --> ({{\{\{}}1,+,16}<%outer.header>,+,2}<%loop> /s 3) U: [-3074457345618258602,3074457345618258603) S: [-3074457345618258602,3074457345618258603) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div3.2 = sdiv i64 %iv.2, 3
+; CHECK-NEXT: --> ({{\{\{}}2,+,16}<%outer.header>,+,2}<%loop> /s 3) U: [-3074457345618258602,3074457345618258603) S: [-3074457345618258602,3074457345618258603) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div3.4 = sdiv i64 %iv.4, 3
+; CHECK-NEXT: --> ({{\{\{}}4,+,16}<%outer.header>,+,2}<%loop> /s 3) U: [-3074457345618258602,3074457345618258603) S: [-3074457345618258602,3074457345618258603) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div3.5 = sdiv i64 %iv.5, 3
+; CHECK-NEXT: --> ({{\{\{}}5,+,16}<%outer.header>,+,2}<%loop> /s 3) U: [-3074457345618258602,3074457345618258603) S: [-3074457345618258602,3074457345618258603) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.next = add i64 %iv, 2
+; CHECK-NEXT: --> {{\{\{}}2,+,16}<%outer.header>,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %outer.iv.next = add i64 %outer.iv, 16
+; CHECK-NEXT: --> {16,+,16}<%outer.header> U: [0,-15) S: [-9223372036854775808,9223372036854775793) Exits: <<Unknown>> LoopDispositions: { %outer.header: Computable, %loop: Invariant }
+; CHECK-NEXT: Determining loop execution counts for: @test_step2_start_outer_add_rec_step_16
+; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+; CHECK-NEXT: Loop %outer.header: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %outer.header: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %outer.header: Unpredictable symbolic max backedge-taken count.
+; CHECK-NEXT: Loop %outer.header: Predicated backedge-taken count is (%m /u 16)
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Equal predicate: (zext i4 (trunc i64 %m to i4) to i64) == 0
+; CHECK-NEXT: Loop %outer.header: Predicated constant max backedge-taken count is i64 1152921504606846975
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Equal predicate: (zext i4 (trunc i64 %m to i4) to i64) == 0
+; CHECK-NEXT: Loop %outer.header: Predicated symbolic max backedge-taken count is (%m /u 16)
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Equal predicate: (zext i4 (trunc i64 %m to i4) to i64) == 0
+;
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+ br label %loop
+
+loop:
+ %iv = phi i64 [ %outer.iv, %outer.header ], [ %iv.next, %loop ]
+ %div.0 = sdiv i64 %iv, 4
+ call void @use(i64 %div.0)
+ %iv.1 = add i64 %iv, 1
+ %div.1 = sdiv i64 %iv.1, 4
+ call void @use(i64 %div.1)
+ %iv.2 = add i64 %iv, 2
+ %div.2 = sdiv i64 %iv.2, 4
+ call void @use(i64 %div.2)
+ %iv.3 = add i64 %iv, 3
+ %div.3 = sdiv i64 %iv.3, 4
+ call void @use(i64 %div.3)
+ %iv.4 = add i64 %iv, 4
+ %div.4 = sdiv i64 %iv.4, 4
+ call void @use(i64 %div.4)
+ %iv.5 = add i64 %iv, 5
+ %div.5 = sdiv i64 %iv.5, 4
+ call void @use(i64 %div.5)
+ %iv.neg.1 = add i64 %iv, -1
+ %div.neg.1 = sdiv i64 %iv.neg.1, 4
+ call void @use(i64 %div.neg.1)
+ %div3.0 = sdiv i64 %iv, 3
+ call void @use(i64 %div3.0)
+ %div3.1 = sdiv i64 %iv.1,3
+ call void @use(i64 %div3.1)
+ %div3.2 = sdiv i64 %iv.2, 3
+ call void @use(i64 %div3.2)
+ %div3.4 = sdiv i64 %iv.4, 3
+ call void @use(i64 %div3.4)
+ %div3.5 = sdiv i64 %iv.5, 3
+ call void @use(i64 %div3.5)
+ %iv.next = add i64 %iv, 2
+ %cond = icmp slt i64 %iv, %n
+ br i1 %cond, label %loop, label %outer.latch
+
+outer.latch:
+ %outer.iv.next = add i64 %outer.iv, 16
+ %outer.ec = icmp eq i64 %outer.iv, %m
+ br i1 %outer.ec, label %exit, label %outer.header
+
+exit:
+ ret void
+}
+
+define void @test_step2_div4_start_outer_add_rec_step_2(i64 %n, i64 %m) {
+; CHECK-LABEL: 'test_step2_div4_start_outer_add_rec_step_2'
+; CHECK-NEXT: Classifying expressions for: @test_step2_div4_start_outer_add_rec_step_2
+; CHECK-NEXT: %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+; CHECK-NEXT: --> {0,+,2}<%outer.header> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %outer.header: Computable, %loop: Invariant }
+; CHECK-NEXT: %iv = phi i64 [ %outer.iv, %outer.header ], [ %iv.next, %loop ]
+; CHECK-NEXT: --> {{\{\{}}0,+,2}<%outer.header>,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.0 = sdiv i64 %iv, 4
+; CHECK-NEXT: --> ({{\{\{}}0,+,2}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.1 = add i64 %iv, 1
+; CHECK-NEXT: --> {{\{\{}}1,+,2}<%outer.header>,+,2}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.1 = sdiv i64 %iv.1, 4
+; CHECK-NEXT: --> ({{\{\{}}1,+,2}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.2 = add i64 %iv, 2
+; CHECK-NEXT: --> {{\{\{}}2,+,2}<%outer.header>,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.2 = sdiv i64 %iv.2, 4
+; CHECK-NEXT: --> ({{\{\{}}2,+,2}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.3 = add i64 %iv, 3
+; CHECK-NEXT: --> {{\{\{}}3,+,2}<%outer.header>,+,2}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.3 = sdiv i64 %iv.3, 4
+; CHECK-NEXT: --> ({{\{\{}}3,+,2}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.4 = add i64 %iv, 4
+; CHECK-NEXT: --> {{\{\{}}4,+,2}<%outer.header>,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.4 = sdiv i64 %iv.4, 4
+; CHECK-NEXT: --> ({{\{\{}}4,+,2}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.5 = add i64 %iv, 5
+; CHECK-NEXT: --> {{\{\{}}5,+,2}<%outer.header>,+,2}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.5 = sdiv i64 %iv.5, 4
+; CHECK-NEXT: --> ({{\{\{}}5,+,2}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.neg.1 = add i64 %iv, -1
+; CHECK-NEXT: --> {{\{\{}}-1,+,2}<%outer.header>,+,2}<%loop> U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div.neg.1 = sdiv i64 %iv.neg.1, 4
+; CHECK-NEXT: --> ({{\{\{}}-1,+,2}<%outer.header>,+,2}<%loop> /s 4) U: [-2305843009213693952,2305843009213693952) S: [-2305843009213693952,2305843009213693952) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div3.0 = sdiv i64 %iv, 3
+; CHECK-NEXT: --> ({{\{\{}}0,+,2}<%outer.header>,+,2}<%loop> /s 3) U: [-3074457345618258602,3074457345618258603) S: [-3074457345618258602,3074457345618258603) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div3.1 = sdiv i64 %iv.1, 3
+; CHECK-NEXT: --> ({{\{\{}}1,+,2}<%outer.header>,+,2}<%loop> /s 3) U: [-3074457345618258602,3074457345618258603) S: [-3074457345618258602,3074457345618258603) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div3.2 = sdiv i64 %iv.2, 3
+; CHECK-NEXT: --> ({{\{\{}}2,+,2}<%outer.header>,+,2}<%loop> /s 3) U: [-3074457345618258602,3074457345618258603) S: [-3074457345618258602,3074457345618258603) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div3.4 = sdiv i64 %iv.4, 3
+; CHECK-NEXT: --> ({{\{\{}}4,+,2}<%outer.header>,+,2}<%loop> /s 3) U: [-3074457345618258602,3074457345618258603) S: [-3074457345618258602,3074457345618258603) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %div3.5 = sdiv i64 %iv.5, 3
+; CHECK-NEXT: --> ({{\{\{}}5,+,2}<%outer.header>,+,2}<%loop> /s 3) U: [-3074457345618258602,3074457345618258603) S: [-3074457345618258602,3074457345618258603) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %iv.next = add i64 %iv, 2
+; CHECK-NEXT: --> {{\{\{}}2,+,2}<%outer.header>,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable, %outer.header: Variant }
+; CHECK-NEXT: %outer.iv.next = add i64 %outer.iv, 2
+; CHECK-NEXT: --> {2,+,2}<%outer.header> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %outer.header: Computable, %loop: Invariant }
+; CHECK-NEXT: Determining loop execution counts for: @test_step2_div4_start_outer_add_rec_step_2
+; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+; CHECK-NEXT: Loop %outer.header: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %outer.header: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %outer.header: Unpredictable symbolic max backedge-taken count.
+; CHECK-NEXT: Loop %outer.header: Predicated backedge-taken count is (%m /u 2)
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Equal predicate: (zext i1 (trunc i64 %m to i1) to i64) == 0
+; CHECK-NEXT: Loop %outer.header: Predicated constant max backedge-taken count is i64 9223372036854775807
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Equal predicate: (zext i1 (trunc i64 %m to i1) to i64) == 0
+; CHECK-NEXT: Loop %outer.header: Predicated symbolic max backedge-taken count is (%m /u 2)
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Equal predicate: (zext i1 (trunc i64 %m to i1) to i64) == 0
+;
+entry:
+ br label %outer.header
+
+outer.header:
+ %outer.iv = phi i64 [ 0, %entry ], [ %outer.iv.next, %outer.latch ]
+ br label %loop
+
+loop:
+ %iv = phi i64 [ %outer.iv, %outer.header ], [ %iv.next, %loop ]
+ %div.0 = sdiv i64 %iv, 4
+ call void @use(i64 %div.0)
+ %iv.1 = add i64 %iv, 1
+ %div.1 = sdiv i64 %iv.1, 4
+ call void @use(i64 %div.1)
+ %iv.2 = add i64 %iv, 2
+ %div.2 = sdiv i64 %iv.2, 4
+ call void @use(i64 %div.2)
+ %iv.3 = add i64 %iv, 3
+ %div.3 = sdiv i64 %iv.3, 4
+ call void @use(i64 %div.3)
+ %iv.4 = add i64 %iv, 4
+ %div.4 = sdiv i64 %iv.4, 4
+ call void @use(i64 %div.4)
+ %iv.5 = add i64 %iv, 5
+ %div.5 = sdiv i64 %iv.5, 4
+ call void @use(i64 %div.5)
+ %iv.neg.1 = add i64 %iv, -1
+ %div.neg.1 = sdiv i64 %iv.neg.1, 4
+ call void @use(i64 %div.neg.1)
+ %div3.0 = sdiv i64 %iv, 3
+ call void @use(i64 %div3.0)
+ %div3.1 = sdiv i64 %iv.1,3
+ call void @use(i64 %div3.1)
+ %div3.2 = sdiv i64 %iv.2, 3
+ call void @use(i64 %div3.2)
+ %div3.4 = sdiv i64 %iv.4, 3
+ call void @use(i64 %div3.4)
+ %div3.5 = sdiv i64 %iv.5, 3
+ call void @use(i64 %div3.5)
+ call void @use(i64 %div.neg.1)
+ %iv.next = add i64 %iv, 2
+ %cond = icmp slt i64 %iv, %n
+ br i1 %cond, label %loop, label %outer.latch
+
+outer.latch:
+ %outer.iv.next = add i64 %outer.iv, 2
+ %outer.ec = icmp eq i64 %outer.iv, %m
+ br i1 %outer.ec, label %exit, label %outer.header
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Analysis/ScalarEvolution/extract-highbits-sameconstmask.ll b/llvm/test/Analysis/ScalarEvolution/extract-highbits-sameconstmask.ll
index 8a5b307367f6e..f45dfbbba66b4 100644
--- a/llvm/test/Analysis/ScalarEvolution/extract-highbits-sameconstmask.ll
+++ b/llvm/test/Analysis/ScalarEvolution/extract-highbits-sameconstmask.ll
@@ -20,9 +20,9 @@ define i32 @sdiv(i32 %val) nounwind {
; CHECK-LABEL: 'sdiv'
; CHECK-NEXT: Classifying expressions for: @sdiv
; CHECK-NEXT: %tmp1 = sdiv i32 %val, 16
-; CHECK-NEXT: --> %tmp1 U: [-134217728,134217728) S: [-134217728,134217728)
+; CHECK-NEXT: --> (%val /s 16) U: [-134217728,134217728) S: [-134217728,134217728)
; CHECK-NEXT: %tmp2 = mul i32 %tmp1, 16
-; CHECK-NEXT: --> (16 * %tmp1)<nsw> U: [0,-15) S: [-2147483648,2147483633)
+; CHECK-NEXT: --> (16 * (%val /s 16))<nsw> U: [0,-15) S: [-2147483648,2147483633)
; CHECK-NEXT: Determining loop execution counts for: @sdiv
;
%tmp1 = sdiv i32 %val, 16
diff --git a/llvm/test/Analysis/ScalarEvolution/extract-highbits-variablemask.ll b/llvm/test/Analysis/ScalarEvolution/extract-highbits-variablemask.ll
index 8461167891e76..ac5a439a2967f 100644
--- a/llvm/test/Analysis/ScalarEvolution/extract-highbits-variablemask.ll
+++ b/llvm/test/Analysis/ScalarEvolution/extract-highbits-variablemask.ll
@@ -22,9 +22,9 @@ define i32 @sdiv(i32 %val, i32 %num) nounwind {
; CHECK-LABEL: 'sdiv'
; CHECK-NEXT: Classifying expressions for: @sdiv
; CHECK-NEXT: %tmp1 = sdiv i32 %val, %num
-; CHECK-NEXT: --> %tmp1 U: full-set S: full-set
+; CHECK-NEXT: --> (%val /s %num) U: full-set S: full-set
; CHECK-NEXT: %tmp2 = mul i32 %tmp1, %num
-; CHECK-NEXT: --> (%num * %tmp1) U: full-set S: full-set
+; CHECK-NEXT: --> ((%val /s %num) * %num) U: full-set S: full-set
; CHECK-NEXT: Determining loop execution counts for: @sdiv
;
%tmp1 = sdiv i32 %val, %num
diff --git a/llvm/test/Analysis/ScalarEvolution/flags-from-poison.ll b/llvm/test/Analysis/ScalarEvolution/flags-from-poison.ll
index ce899726d44cd..d07909162cbf5 100644
--- a/llvm/test/Analysis/ScalarEvolution/flags-from-poison.ll
+++ b/llvm/test/Analysis/ScalarEvolution/flags-from-poison.ll
@@ -971,7 +971,7 @@ define void @test-add-div(ptr %input, i32 %offset, i32 %numIterations) {
; CHECK-NEXT: %j = add nsw i32 %i, %offset
; CHECK-NEXT: --> {%offset,+,1}<nsw><%loop> U: full-set S: full-set Exits: (-1 + %offset + %numIterations) LoopDispositions: { %loop: Computable }
; CHECK-NEXT: %q = sdiv i32 %numIterations, %j
-; CHECK-NEXT: --> %q U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Variant }
+; CHECK-NEXT: --> (%numIterations /s {%offset,+,1}<nsw><%loop>) U: full-set S: full-set Exits: (%numIterations /s (-1 + %offset + %numIterations)) LoopDispositions: { %loop: Computable }
; CHECK-NEXT: %nexti = add nsw i32 %i, 1
; CHECK-NEXT: --> {1,+,1}<nuw><nsw><%loop> U: [1,-2147483648) S: [1,-2147483648) Exits: %numIterations LoopDispositions: { %loop: Computable }
; CHECK-NEXT: Determining loop execution counts for: @test-add-div
@@ -1004,7 +1004,7 @@ define void @test-add-div2(ptr %input, i32 %offset, i32 %numIterations) {
; CHECK-NEXT: %j = add nsw i32 %i, %offset
; CHECK-NEXT: --> {%offset,+,1}<nw><%loop> U: full-set S: full-set Exits: (-1 + %offset + %numIterations) LoopDispositions: { %loop: Computable }
; CHECK-NEXT: %q = sdiv i32 %j, %numIterations
-; CHECK-NEXT: --> %q U: full-set S: full-set Exits: <<Unknown>> LoopDispositions: { %loop: Variant }
+; CHECK-NEXT: --> ({%offset,+,1}<nw><%loop> /s %numIterations) U: full-set S: full-set Exits: ((-1 + %offset + %numIterations) /s %numIterations) LoopDispositions: { %loop: Computable }
; CHECK-NEXT: %nexti = add nsw i32 %i, 1
; CHECK-NEXT: --> {1,+,1}<nuw><nsw><%loop> U: [1,-2147483648) S: [1,-2147483648) Exits: %numIterations LoopDispositions: { %loop: Computable }
; CHECK-NEXT: Determining loop execution counts for: @test-add-div2
diff --git a/llvm/test/Analysis/ScalarEvolution/implied-via-division.ll b/llvm/test/Analysis/ScalarEvolution/implied-via-division.ll
index d83301243ef30..733ad98f4e3df 100644
--- a/llvm/test/Analysis/ScalarEvolution/implied-via-division.ll
+++ b/llvm/test/Analysis/ScalarEvolution/implied-via-division.ll
@@ -6,9 +6,9 @@ define void @implied1(i32 %n) {
; Prove that (n s> 1) ===> (n / 2 s> 0).
; CHECK-LABEL: 'implied1'
; CHECK-NEXT: Determining loop execution counts for: @implied1
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + %n.div.2)<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741822
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + %n.div.2)<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -31,9 +31,9 @@ define void @implied1_samesign(i32 %n) {
; Prove that (n > 1) ===> (n / 2 s> 0).
; CHECK-LABEL: 'implied1_samesign'
; CHECK-NEXT: Determining loop execution counts for: @implied1_samesign
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + %n.div.2)<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax (%n /s 2)))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741822
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + %n.div.2)<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax (%n /s 2)))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -56,9 +56,9 @@ define void @implied1_neg(i32 %n) {
; Prove that (n s> 0) =\=> (n / 2 s> 0).
; CHECK-LABEL: 'implied1_neg'
; CHECK-NEXT: Determining loop execution counts for: @implied1_neg
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax %n.div.2))<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax (%n /s 2)))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741822
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax %n.div.2))<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax (%n /s 2)))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -81,9 +81,9 @@ define void @implied2(i32 %n) {
; Prove that (n s>= 2) ===> (n / 2 s> 0).
; CHECK-LABEL: 'implied2'
; CHECK-NEXT: Determining loop execution counts for: @implied2
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + %n.div.2)<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741822
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + %n.div.2)<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -106,9 +106,9 @@ define void @implied2_samesign(i32 %n) {
; Prove that (n >= 2) ===> (n / 2 s> 0).
; CHECK-LABEL: 'implied2_samesign'
; CHECK-NEXT: Determining loop execution counts for: @implied2_samesign
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax %n.div.2))<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax (%n /s 2)))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741822
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax %n.div.2))<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax (%n /s 2)))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -131,9 +131,9 @@ define void @implied2_neg(i32 %n) {
; Prove that (n s>= 1) =\=> (n / 2 s> 0).
; CHECK-LABEL: 'implied2_neg'
; CHECK-NEXT: Determining loop execution counts for: @implied2_neg
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax %n.div.2))<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax (%n /s 2)))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741822
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax %n.div.2))<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax (%n /s 2)))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -156,9 +156,9 @@ define void @implied3(i32 %n) {
; Prove that (n s> -2) ===> (n / 2 s>= 0).
; CHECK-LABEL: 'implied3'
; CHECK-NEXT: Determining loop execution counts for: @implied3
-; CHECK-NEXT: Loop %header: backedge-taken count is (1 + %n.div.2)<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741824
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + %n.div.2)<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -181,10 +181,10 @@ define void @implied3_samesign(i32 %n) {
; Prove that (n > -2) ===> (n / 2 s>= 0).
; CHECK-LABEL: 'implied3_samesign'
; CHECK-NEXT: Determining loop execution counts for: @implied3_samesign
-; CHECK-NEXT: Loop %header: backedge-taken count is (1 + %n.div.2)<nsw>
-; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + %n.div.2)<nsw>
-; CHECK-NEXT: Loop %header: Trip multiple is 1
+; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (%n /s 2))<nsw>
+; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741824
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (%n /s 2))<nsw>
+; CHECK-NEXT: Loop %header: Trip multiple is 2
;
entry:
%cmp1 = icmp samesign ugt i32 %n, -2
@@ -206,9 +206,9 @@ define void @implied3_neg(i32 %n) {
; Prove that (n > -3) =\=> (n / 2 >= 0).
; CHECK-LABEL: 'implied3_neg'
; CHECK-NEXT: Determining loop execution counts for: @implied3_neg
-; CHECK-NEXT: Loop %header: backedge-taken count is (0 smax (1 + %n.div.2)<nsw>)
+; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741824
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (0 smax (1 + %n.div.2)<nsw>)
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -231,9 +231,9 @@ define void @implied4(i32 %n) {
; Prove that (n s>= -1) ===> (n / 2 s>= 0).
; CHECK-LABEL: 'implied4'
; CHECK-NEXT: Determining loop execution counts for: @implied4
-; CHECK-NEXT: Loop %header: backedge-taken count is (1 + %n.div.2)<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741824
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + %n.div.2)<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -256,10 +256,10 @@ define void @implied4_samesign(i32 %n) {
; Prove that (n >= -1) ===> (n / 2 s>= 0).
; CHECK-LABEL: 'implied4_samesign'
; CHECK-NEXT: Determining loop execution counts for: @implied4_samesign
-; CHECK-NEXT: Loop %header: backedge-taken count is (1 + %n.div.2)<nsw>
-; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + %n.div.2)<nsw>
-; CHECK-NEXT: Loop %header: Trip multiple is 1
+; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (%n /s 2))<nsw>
+; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741824
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (%n /s 2))<nsw>
+; CHECK-NEXT: Loop %header: Trip multiple is 2
;
entry:
%cmp1 = icmp samesign uge i32 %n, -1
@@ -281,9 +281,9 @@ define void @implied4_neg(i32 %n) {
; Prove that (n s>= -2) =\=> (n / 2 s>= 0).
; CHECK-LABEL: 'implied4_neg'
; CHECK-NEXT: Determining loop execution counts for: @implied4_neg
-; CHECK-NEXT: Loop %header: backedge-taken count is (0 smax (1 + %n.div.2)<nsw>)
+; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741824
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (0 smax (1 + %n.div.2)<nsw>)
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -306,9 +306,9 @@ define void @test_ext_01(i32 %n) nounwind {
; Prove that (n > 1) ===> (n / 2 > 0).
; CHECK-LABEL: 'test_ext_01'
; CHECK-NEXT: Determining loop execution counts for: @test_ext_01
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (sext i32 %n.div.2 to i64))<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i64 1073741822
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (sext i32 %n.div.2 to i64))<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -332,9 +332,9 @@ define void @test_ext_01neg(i32 %n) nounwind {
; Prove that (n > 0) =\=> (n / 2 > 0).
; CHECK-LABEL: 'test_ext_01neg'
; CHECK-NEXT: Determining loop execution counts for: @test_ext_01neg
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax (sext i32 %n.div.2 to i64)))<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax (sext i32 (%n /s 2) to i64)))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i64 1073741822
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax (sext i32 %n.div.2 to i64)))<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax (sext i32 (%n /s 2) to i64)))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -358,9 +358,9 @@ define void @test_ext_02(i32 %n) nounwind {
; Prove that (n >= 2) ===> (n / 2 > 0).
; CHECK-LABEL: 'test_ext_02'
; CHECK-NEXT: Determining loop execution counts for: @test_ext_02
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (sext i32 %n.div.2 to i64))<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i64 1073741822
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (sext i32 %n.div.2 to i64))<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -384,9 +384,9 @@ define void @test_ext_02neg(i32 %n) nounwind {
; Prove that (n >= 1) =\=> (n / 2 > 0).
; CHECK-LABEL: 'test_ext_02neg'
; CHECK-NEXT: Determining loop execution counts for: @test_ext_02neg
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax (sext i32 %n.div.2 to i64)))<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax (sext i32 (%n /s 2) to i64)))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i64 1073741822
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax (sext i32 %n.div.2 to i64)))<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax (sext i32 (%n /s 2) to i64)))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -410,9 +410,9 @@ define void @test_ext_03(i32 %n) nounwind {
; Prove that (n > -2) ===> (n / 2 >= 0).
; CHECK-LABEL: 'test_ext_03'
; CHECK-NEXT: Determining loop execution counts for: @test_ext_03
-; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (sext i32 %n.div.2 to i64))<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i64 1073741824
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (sext i32 %n.div.2 to i64))<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -436,9 +436,9 @@ define void @test_ext_03neg(i32 %n) nounwind {
; Prove that (n > -3) =\=> (n / 2 >= 0).
; CHECK-LABEL: 'test_ext_03neg'
; CHECK-NEXT: Determining loop execution counts for: @test_ext_03neg
-; CHECK-NEXT: Loop %header: backedge-taken count is (0 smax (1 + (sext i32 %n.div.2 to i64))<nsw>)
+; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i64 1073741824
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (0 smax (1 + (sext i32 %n.div.2 to i64))<nsw>)
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -462,9 +462,9 @@ define void @test_ext_04(i32 %n) nounwind {
; Prove that (n >= -1) ===> (n / 2 >= 0).
; CHECK-LABEL: 'test_ext_04'
; CHECK-NEXT: Determining loop execution counts for: @test_ext_04
-; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (sext i32 %n.div.2 to i64))<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i64 1073741824
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (sext i32 %n.div.2 to i64))<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -488,9 +488,9 @@ define void @test_ext_04neg(i32 %n) nounwind {
; Prove that (n >= -2) =\=> (n / 2 >= 0).
; CHECK-LABEL: 'test_ext_04neg'
; CHECK-NEXT: Determining loop execution counts for: @test_ext_04neg
-; CHECK-NEXT: Loop %header: backedge-taken count is (0 smax (1 + (sext i32 %n.div.2 to i64))<nsw>)
+; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i64 1073741824
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (0 smax (1 + (sext i32 %n.div.2 to i64))<nsw>)
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (sext i32 (%n /s 2) to i64))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -514,9 +514,9 @@ define void @swapped_predicate(i32 %n) {
; Prove that (n s>= 1) ===> (0 s>= -n / 2).
; CHECK-LABEL: 'swapped_predicate'
; CHECK-NEXT: Determining loop execution counts for: @swapped_predicate
-; CHECK-NEXT: Loop %header: backedge-taken count is (1 + %n.div.2)<nuw><nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 * (0 smin (-1 + (-1 * (%n /s 2))<nsw>)<nsw>))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741824
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + %n.div.2)<nuw><nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 * (0 smin (-1 + (-1 * (%n /s 2))<nsw>)<nsw>))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
diff --git a/llvm/test/Analysis/ScalarEvolution/mul-sdiv-folds.ll b/llvm/test/Analysis/ScalarEvolution/mul-sdiv-folds.ll
new file mode 100644
index 0000000000000..5bae7557c8535
--- /dev/null
+++ b/llvm/test/Analysis/ScalarEvolution/mul-sdiv-folds.ll
@@ -0,0 +1,145 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -passes='print<scalar-evolution>' -disable-output %s 2>&1 | FileCheck %s
+
+declare void @use(ptr)
+
+define void @sdiv3_and_sdiv5_mul_4(i1 %c, ptr %A) {
+; CHECK-LABEL: 'sdiv3_and_sdiv5_mul_4'
+; CHECK-NEXT: Classifying expressions for: @sdiv3_and_sdiv5_mul_4
+; CHECK-NEXT: %start = select i1 %c, i32 512, i32 0
+; CHECK-NEXT: --> %start U: [0,513) S: [0,513)
+; CHECK-NEXT: %div.3 = sdiv i32 %start, -3
+; CHECK-NEXT: --> (%start /s -3) U: [-170,1) S: [-170,1)
+; CHECK-NEXT: %div.5 = sdiv i32 %start, -5
+; CHECK-NEXT: --> (%start /s -5) U: [-102,1) S: [-102,1)
+; CHECK-NEXT: %iv.start = zext i32 %div.5 to i64
+; CHECK-NEXT: --> (zext i32 (%start /s -5) to i64) U: [0,4294967296) S: [0,4294967296)
+; CHECK-NEXT: %wide.trip.count = zext i32 %div.3 to i64
+; CHECK-NEXT: --> (zext i32 (%start /s -3) to i64) U: [0,4294967296) S: [0,4294967296)
+; CHECK-NEXT: %iv = phi i64 [ %iv.start, %entry ], [ %iv.next, %loop ]
+; CHECK-NEXT: --> {(zext i32 (%start /s -5) to i64),+,1}<%loop> U: full-set S: full-set Exits: (zext i32 (%start /s -3) to i64) LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %gep.8 = getelementptr i8, ptr %A, i64 %iv
+; CHECK-NEXT: --> {((zext i32 (%start /s -5) to i64) + %A),+,1}<%loop> U: full-set S: full-set Exits: ((zext i32 (%start /s -3) to i64) + %A) LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %gep.16 = getelementptr i16, ptr %A, i64 %iv
+; CHECK-NEXT: --> {((2 * (zext i32 (%start /s -5) to i64))<nuw><nsw> + %A),+,2}<%loop> U: full-set S: full-set Exits: ((2 * (zext i32 (%start /s -3) to i64))<nuw><nsw> + %A) LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %gep.32 = getelementptr i32, ptr %A, i64 %iv
+; CHECK-NEXT: --> {((4 * (zext i32 (%start /s -5) to i64))<nuw><nsw> + %A),+,4}<%loop> U: full-set S: full-set Exits: ((4 * (zext i32 (%start /s -3) to i64))<nuw><nsw> + %A) LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %gep.40 = getelementptr <{ i32, i8 }>, ptr %A, i64 %iv
+; CHECK-NEXT: --> {((5 * (zext i32 (%start /s -5) to i64))<nuw><nsw> + %A),+,5}<%loop> U: full-set S: full-set Exits: ((5 * (zext i32 (%start /s -3) to i64))<nuw><nsw> + %A) LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %gep.48 = getelementptr <{ i32, i16 }>, ptr %A, i64 %iv
+; CHECK-NEXT: --> {((6 * (zext i32 (%start /s -5) to i64))<nuw><nsw> + %A),+,6}<%loop> U: full-set S: full-set Exits: ((6 * (zext i32 (%start /s -3) to i64))<nuw><nsw> + %A) LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.next = add i64 %iv, 1
+; CHECK-NEXT: --> {(1 + (zext i32 (%start /s -5) to i64))<nuw><nsw>,+,1}<%loop> U: full-set S: full-set Exits: (1 + (zext i32 (%start /s -3) to i64))<nuw><nsw> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: Determining loop execution counts for: @sdiv3_and_sdiv5_mul_4
+; CHECK-NEXT: Loop %loop: backedge-taken count is ((zext i32 (%start /s -3) to i64) + (-1 * (zext i32 (%start /s -5) to i64))<nsw>)
+; CHECK-NEXT: Loop %loop: constant max backedge-taken count is i64 -1
+; CHECK-NEXT: Loop %loop: symbolic max backedge-taken count is ((zext i32 (%start /s -3) to i64) + (-1 * (zext i32 (%start /s -5) to i64))<nsw>)
+; CHECK-NEXT: Loop %loop: Trip multiple is 1
+;
+entry:
+ %start = select i1 %c, i32 512, i32 0
+ %div.3 = sdiv i32 %start, -3
+ %div.5 = sdiv i32 %start, -5
+ %iv.start = zext i32 %div.5 to i64
+ %wide.trip.count = zext i32 %div.3 to i64
+ br label %loop
+
+loop:
+ %iv = phi i64 [ %iv.start, %entry ], [ %iv.next, %loop ]
+ %gep.8 = getelementptr i8, ptr %A, i64 %iv
+ call void @use(ptr %gep.8)
+ %gep.16 = getelementptr i16, ptr %A, i64 %iv
+ call void @use(ptr %gep.16)
+ %gep.32 = getelementptr i32, ptr %A, i64 %iv
+ call void @use(ptr %gep.32)
+ %gep.40 = getelementptr <{i32, i8}>, ptr %A, i64 %iv
+ call void @use(ptr %gep.40)
+ %gep.48 = getelementptr <{i32, i16}>, ptr %A, i64 %iv
+ call void @use(ptr %gep.48)
+ %iv.next = add i64 %iv, 1
+ %ec = icmp eq i64 %iv, %wide.trip.count
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+declare void @use.i64(i64)
+
+define void @btc_depends_on_div_mul(i64 %x) {
+; CHECK-LABEL: 'btc_depends_on_div_mul'
+; CHECK-NEXT: Classifying expressions for: @btc_depends_on_div_mul
+; CHECK-NEXT: %div.16 = sdiv i64 %x, -16
+; CHECK-NEXT: --> (%x /s -16) U: [-576460752303423487,576460752303423489) S: [-576460752303423487,576460752303423489)
+; CHECK-NEXT: %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+; CHECK-NEXT: --> {0,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: %iv.next = add i64 %iv, 2
+; CHECK-NEXT: --> {2,+,2}<%loop> U: [0,-1) S: [-9223372036854775808,9223372036854775807) Exits: <<Unknown>> LoopDispositions: { %loop: Computable }
+; CHECK-NEXT: Determining loop execution counts for: @btc_depends_on_div_mul
+; CHECK-NEXT: Loop %loop: Unpredictable backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable constant max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Unpredictable symbolic max backedge-taken count.
+; CHECK-NEXT: Loop %loop: Predicated backedge-taken count is ((-2 + (%x /s -16))<nsw> /u 2)
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Equal predicate: (zext i1 (trunc i64 (%x /s -16) to i1) to i64) == 0
+; CHECK-NEXT: Loop %loop: Predicated constant max backedge-taken count is i64 9223372036854775807
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Equal predicate: (zext i1 (trunc i64 (%x /s -16) to i1) to i64) == 0
+; CHECK-NEXT: Loop %loop: Predicated symbolic max backedge-taken count is ((-2 + (%x /s -16))<nsw> /u 2)
+; CHECK-NEXT: Predicates:
+; CHECK-NEXT: Equal predicate: (zext i1 (trunc i64 (%x /s -16) to i1) to i64) == 0
+;
+entry:
+ %div.16 = sdiv i64 %x, -16
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ call void @use.i64(i64 %iv)
+ %iv.next = add i64 %iv, 2
+ %ec = icmp eq i64 %iv.next, %div.16
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+define noundef i64 @sdiv_mul_common_vscale_factor(i64 %a, i64 %b) {
+; CHECK-LABEL: 'sdiv_mul_common_vscale_factor'
+; CHECK-NEXT: Classifying expressions for: @sdiv_mul_common_vscale_factor
+; CHECK-NEXT: %vs = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: --> vscale U: [1,0) S: [1,0)
+; CHECK-NEXT: %a.vs = mul i64 %a, %vs
+; CHECK-NEXT: --> (vscale * %a) U: full-set S: full-set
+; CHECK-NEXT: %b.vs = mul i64 %b, %vs
+; CHECK-NEXT: --> (vscale * %b) U: full-set S: full-set
+; CHECK-NEXT: %div = sdiv i64 %a.vs, %b.vs
+; CHECK-NEXT: --> ((vscale * %a) /s (vscale * %b)) U: full-set S: full-set
+; CHECK-NEXT: Determining loop execution counts for: @sdiv_mul_common_vscale_factor
+;
+ %vs = call i64 @llvm.vscale()
+ %a.vs = mul i64 %a, %vs
+ %b.vs = mul i64 %b, %vs
+ %div = sdiv i64 %a.vs, %b.vs
+ ret i64 %div
+}
+
+define noundef i64 @sdiv_mul_nuw_common_vscale_factor(i64 %a, i64 %b) {
+; CHECK-LABEL: 'sdiv_mul_nuw_common_vscale_factor'
+; CHECK-NEXT: Classifying expressions for: @sdiv_mul_nuw_common_vscale_factor
+; CHECK-NEXT: %vs = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: --> vscale U: [1,0) S: [1,0)
+; CHECK-NEXT: %a.vs = mul nsw i64 %a, %vs
+; CHECK-NEXT: --> (vscale * %a)<nsw> U: full-set S: full-set
+; CHECK-NEXT: %b.vs = mul nsw i64 %b, %vs
+; CHECK-NEXT: --> (vscale * %b)<nsw> U: full-set S: full-set
+; CHECK-NEXT: %div = sdiv i64 %a.vs, %b.vs
+; CHECK-NEXT: --> (%a /s %b) U: full-set S: full-set
+; CHECK-NEXT: Determining loop execution counts for: @sdiv_mul_nuw_common_vscale_factor
+;
+ %vs = call i64 @llvm.vscale()
+ %a.vs = mul nsw i64 %a, %vs
+ %b.vs = mul nsw i64 %b, %vs
+ %div = sdiv i64 %a.vs, %b.vs
+ ret i64 %div
+}
diff --git a/llvm/test/Analysis/ScalarEvolution/ptrtoint-constantexpr-loop.ll b/llvm/test/Analysis/ScalarEvolution/ptrtoint-constantexpr-loop.ll
index 99ecb3c9a7cb9..664a4ba4df60e 100644
--- a/llvm/test/Analysis/ScalarEvolution/ptrtoint-constantexpr-loop.ll
+++ b/llvm/test/Analysis/ScalarEvolution/ptrtoint-constantexpr-loop.ll
@@ -327,7 +327,7 @@ define i64 @sext_like_noop(i32 %n) {
; PTR64_IDX64-NEXT: %ii = sext i32 %i to i64
; PTR64_IDX64-NEXT: --> (sext i32 {1,+,1}<nuw><%for.body> to i64) U: [-2147483648,2147483648) S: [-2147483648,2147483648) --> (sext i32 (-1 + ptrtoint (ptr @sext_like_noop to i32)) to i64) U: [-2147483648,2147483648) S: [-2147483648,2147483648)
; PTR64_IDX64-NEXT: %div = sdiv i64 55555, %ii
-; PTR64_IDX64-NEXT: --> %div U: full-set S: full-set
+; PTR64_IDX64-NEXT: --> (55555 /s (sext i32 {1,+,1}<nuw><%for.body> to i64)) U: [-55555,55556) S: [-55555,55556) --> (55555 /s (sext i32 (-1 + ptrtoint (ptr @sext_like_noop to i32)) to i64)) U: [-55555,55556) S: [-55555,55556)
; PTR64_IDX64-NEXT: %i = phi i32 [ %inc, %for.body ], [ 1, %entry ]
; PTR64_IDX64-NEXT: --> {1,+,1}<nuw><%for.body> U: [1,0) S: [1,0) Exits: (-1 + ptrtoint (ptr @sext_like_noop to i32)) LoopDispositions: { %for.body: Computable }
; PTR64_IDX64-NEXT: %inc = add nuw i32 %i, 1
@@ -343,7 +343,7 @@ define i64 @sext_like_noop(i32 %n) {
; PTR64_IDX32-NEXT: %ii = sext i32 %i to i64
; PTR64_IDX32-NEXT: --> (sext i32 {1,+,1}<nuw><%for.body> to i64) U: [-2147483648,2147483648) S: [-2147483648,2147483648) --> (sext i32 (-1 + ptrtoint (ptr @sext_like_noop to i32)) to i64) U: [-2147483648,2147483648) S: [-2147483648,2147483648)
; PTR64_IDX32-NEXT: %div = sdiv i64 55555, %ii
-; PTR64_IDX32-NEXT: --> %div U: full-set S: full-set
+; PTR64_IDX32-NEXT: --> (55555 /s (sext i32 {1,+,1}<nuw><%for.body> to i64)) U: [-55555,55556) S: [-55555,55556) --> (55555 /s (sext i32 (-1 + ptrtoint (ptr @sext_like_noop to i32)) to i64)) U: [-55555,55556) S: [-55555,55556)
; PTR64_IDX32-NEXT: %i = phi i32 [ %inc, %for.body ], [ 1, %entry ]
; PTR64_IDX32-NEXT: --> {1,+,1}<nuw><%for.body> U: [1,0) S: [1,0) Exits: (-1 + ptrtoint (ptr @sext_like_noop to i32)) LoopDispositions: { %for.body: Computable }
; PTR64_IDX32-NEXT: %inc = add nuw i32 %i, 1
@@ -359,7 +359,7 @@ define i64 @sext_like_noop(i32 %n) {
; PTR16_IDX16-NEXT: %ii = sext i32 %i to i64
; PTR16_IDX16-NEXT: --> (sext i32 {1,+,1}<nuw><%for.body> to i64) U: [-2147483648,2147483648) S: [-2147483648,2147483648) --> (-1 + (zext i32 ptrtoint (ptr @sext_like_noop to i32) to i64))<nsw> U: [-1,65535) S: [-1,65535)
; PTR16_IDX16-NEXT: %div = sdiv i64 55555, %ii
-; PTR16_IDX16-NEXT: --> %div U: full-set S: full-set
+; PTR16_IDX16-NEXT: --> (55555 /s (sext i32 {1,+,1}<nuw><%for.body> to i64)) U: [-55555,55556) S: [-55555,55556) --> (55555 /s (-1 + (zext i32 ptrtoint (ptr @sext_like_noop to i32) to i64))<nsw>) U: [-55555,55556) S: [-55555,55556)
; PTR16_IDX16-NEXT: %i = phi i32 [ %inc, %for.body ], [ 1, %entry ]
; PTR16_IDX16-NEXT: --> {1,+,1}<nuw><%for.body> U: [1,0) S: [1,0) Exits: (-1 + ptrtoint (ptr @sext_like_noop to i32))<nsw> LoopDispositions: { %for.body: Computable }
; PTR16_IDX16-NEXT: %inc = add nuw i32 %i, 1
diff --git a/llvm/test/Analysis/ScalarEvolution/sdiv.ll b/llvm/test/Analysis/ScalarEvolution/sdiv.ll
index acc6ab01978f8..7ef90ccb4e458 100644
--- a/llvm/test/Analysis/ScalarEvolution/sdiv.ll
+++ b/llvm/test/Analysis/ScalarEvolution/sdiv.ll
@@ -4,6 +4,255 @@
target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-f80:128-n8:16:32:64-S128"
target triple = "x86_64-unknown-linux-gnu"
+declare void @noundef(i8 noundef)
+
+define i8 @zero(i8 %x) {
+; CHECK-LABEL: 'zero'
+; CHECK-NEXT: Classifying expressions for: @zero
+; CHECK-NEXT: %div = sdiv i8 0, %x
+; CHECK-NEXT: --> 0 U: [0,1) S: [0,1)
+; CHECK-NEXT: Determining loop execution counts for: @zero
+;
+ %div = sdiv i8 0, %x
+ ret i8 %div
+}
+
+define i8 @by_one(i8 %x) {
+; CHECK-LABEL: 'by_one'
+; CHECK-NEXT: Classifying expressions for: @by_one
+; CHECK-NEXT: %div = sdiv i8 %x, 1
+; CHECK-NEXT: --> %x U: full-set S: full-set
+; CHECK-NEXT: Determining loop execution counts for: @by_one
+;
+ %div = sdiv i8 %x, 1
+ ret i8 %div
+}
+
+define i8 @by_allones_smin(i8 %x) {
+; CHECK-LABEL: 'by_allones_smin'
+; CHECK-NEXT: Classifying expressions for: @by_allones_smin
+; CHECK-NEXT: %div = sdiv i8 %x, -1
+; CHECK-NEXT: --> (%x /s -1) U: [-127,-128) S: [-127,-128)
+; CHECK-NEXT: Determining loop execution counts for: @by_allones_smin
+;
+ %div = sdiv i8 %x, -1
+ ret i8 %div
+}
+
+define i8 @by_allones_no_smin(i8 range(i8 -16, 0) %x) {
+; CHECK-LABEL: 'by_allones_no_smin'
+; CHECK-NEXT: Classifying expressions for: @by_allones_no_smin
+; CHECK-NEXT: %div = sdiv i8 %x, -1
+; CHECK-NEXT: --> (-1 * %x)<nsw> U: [1,17) S: [1,17)
+; CHECK-NEXT: Determining loop execution counts for: @by_allones_no_smin
+;
+ %div = sdiv i8 %x, -1
+ ret i8 %div
+}
+
+define i8 @mul_nsw_const_by_const1(i8 range(i8 -16, 0) %x) {
+; CHECK-LABEL: 'mul_nsw_const_by_const1'
+; CHECK-NEXT: Classifying expressions for: @mul_nsw_const_by_const1
+; CHECK-NEXT: %mul = mul i8 %x, -3
+; CHECK-NEXT: --> (-3 * %x)<nsw> U: [3,49) S: [3,49)
+; CHECK-NEXT: %div = sdiv i8 %mul, -3
+; CHECK-NEXT: --> %x U: [-16,0) S: [-16,0)
+; CHECK-NEXT: Determining loop execution counts for: @mul_nsw_const_by_const1
+;
+ %mul = mul i8 %x, -3
+ %div = sdiv i8 %mul, -3
+ ret i8 %div
+}
+
+define i8 @mul_nsw_const_by_const2(i8 range(i8 -16, 0) %x) {
+; CHECK-LABEL: 'mul_nsw_const_by_const2'
+; CHECK-NEXT: Classifying expressions for: @mul_nsw_const_by_const2
+; CHECK-NEXT: %mul = mul i8 %x, -6
+; CHECK-NEXT: --> (-6 * %x)<nsw> U: [6,97) S: [6,97)
+; CHECK-NEXT: %div = sdiv i8 %mul, -3
+; CHECK-NEXT: --> (2 * %x)<nsw> U: [-32,-1) S: [-32,-1)
+; CHECK-NEXT: Determining loop execution counts for: @mul_nsw_const_by_const2
+;
+ %mul = mul i8 %x, -6
+ %div = sdiv i8 %mul, -3
+ ret i8 %div
+}
+
+define i8 @mul_nsw_const_by_const_common_factor(i8 range(i8 -16, 0) %x) {
+; CHECK-LABEL: 'mul_nsw_const_by_const_common_factor'
+; CHECK-NEXT: Classifying expressions for: @mul_nsw_const_by_const_common_factor
+; CHECK-NEXT: %mul = mul i8 %x, -6
+; CHECK-NEXT: --> (-6 * %x)<nsw> U: [6,97) S: [6,97)
+; CHECK-NEXT: %div = sdiv i8 %mul, -4
+; CHECK-NEXT: --> ((-3 * %x)<nsw> /s -2) U: [-24,0) S: [-24,0)
+; CHECK-NEXT: Determining loop execution counts for: @mul_nsw_const_by_const_common_factor
+;
+ %mul = mul i8 %x, -6
+ %div = sdiv i8 %mul, -4
+ ret i8 %div
+}
+
+define i8 @mul_nsw_const_by_const_no_common_factor(i8 range(i8 -16, 0) %x) {
+; CHECK-LABEL: 'mul_nsw_const_by_const_no_common_factor'
+; CHECK-NEXT: Classifying expressions for: @mul_nsw_const_by_const_no_common_factor
+; CHECK-NEXT: %mul = mul i8 %x, -7
+; CHECK-NEXT: --> (-7 * %x)<nsw> U: [7,113) S: [7,113)
+; CHECK-NEXT: %div = sdiv i8 %mul, -4
+; CHECK-NEXT: --> ((-7 * %x)<nsw> /s -4) U: [-28,0) S: [-28,0)
+; CHECK-NEXT: Determining loop execution counts for: @mul_nsw_const_by_const_no_common_factor
+;
+ %mul = mul i8 %x, -7
+ %div = sdiv i8 %mul, -4
+ ret i8 %div
+}
+
+define i8 @mul_const_by_const_not_nsw(i8 %x) {
+; CHECK-LABEL: 'mul_const_by_const_not_nsw'
+; CHECK-NEXT: Classifying expressions for: @mul_const_by_const_not_nsw
+; CHECK-NEXT: %mul = mul i8 %x, -3
+; CHECK-NEXT: --> (-3 * %x) U: full-set S: full-set
+; CHECK-NEXT: %div = sdiv i8 %mul, -3
+; CHECK-NEXT: --> ((-3 * %x) /s -3) U: [-42,43) S: [-42,43)
+; CHECK-NEXT: Determining loop execution counts for: @mul_const_by_const_not_nsw
+;
+ %mul = mul i8 %x, -3
+ %div = sdiv i8 %mul, -3
+ ret i8 %div
+}
+
+define i8 @mul_nsw_by_factor(i8 %x, i8 %y) {
+; CHECK-LABEL: 'mul_nsw_by_factor'
+; CHECK-NEXT: Classifying expressions for: @mul_nsw_by_factor
+; CHECK-NEXT: %mul = mul nsw i8 %x, %y
+; CHECK-NEXT: --> (%x * %y)<nsw> U: full-set S: full-set
+; CHECK-NEXT: %div = sdiv i8 %mul, %y
+; CHECK-NEXT: --> %x U: full-set S: full-set
+; CHECK-NEXT: Determining loop execution counts for: @mul_nsw_by_factor
+;
+ %mul = mul nsw i8 %x, %y
+ call void @noundef(i8 %mul)
+ %div = sdiv i8 %mul, %y
+ ret i8 %div
+}
+
+define i8 @mul_by_factor_not_nsw(i8 %x, i8 %y) {
+; CHECK-LABEL: 'mul_by_factor_not_nsw'
+; CHECK-NEXT: Classifying expressions for: @mul_by_factor_not_nsw
+; CHECK-NEXT: %mul = mul i8 %x, %y
+; CHECK-NEXT: --> (%x * %y) U: full-set S: full-set
+; CHECK-NEXT: %div = sdiv i8 %mul, %y
+; CHECK-NEXT: --> ((%x * %y) /s %y) U: full-set S: full-set
+; CHECK-NEXT: Determining loop execution counts for: @mul_by_factor_not_nsw
+;
+ %mul = mul i8 %x, %y
+ %div = sdiv i8 %mul, %y
+ ret i8 %div
+}
+
+define i8 @div_div_fold(i8 %x) {
+; CHECK-LABEL: 'div_div_fold'
+; CHECK-NEXT: Classifying expressions for: @div_div_fold
+; CHECK-NEXT: %div = sdiv i8 %x, -3
+; CHECK-NEXT: --> (%x /s -3) U: [-42,43) S: [-42,43)
+; CHECK-NEXT: %div.2 = sdiv i8 %div, -5
+; CHECK-NEXT: --> (%x /s 15) U: [-8,9) S: [-8,9)
+; CHECK-NEXT: Determining loop execution counts for: @div_div_fold
+;
+ %div = sdiv i8 %x, -3
+ %div.2 = sdiv i8 %div, -5
+ ret i8 %div.2
+}
+
+define i8 @add_nsw_distribute_fold(i8 range(i8 -16, 0) %x) {
+; CHECK-LABEL: 'add_nsw_distribute_fold'
+; CHECK-NEXT: Classifying expressions for: @add_nsw_distribute_fold
+; CHECK-NEXT: %div = add i8 %x, -2
+; CHECK-NEXT: --> (-2 + %x)<nsw> U: [-18,-2) S: [-18,-2)
+; CHECK-NEXT: %div.2 = sdiv i8 %div, -1
+; CHECK-NEXT: --> (2 + (-1 * %x)<nsw>)<nuw><nsw> U: [3,19) S: [3,19)
+; CHECK-NEXT: Determining loop execution counts for: @add_nsw_distribute_fold
+;
+ %div = add i8 %x, -2
+ %div.2 = sdiv i8 %div, -1
+ ret i8 %div.2
+}
+
+define i8 @add_no_nsw_distribute_fold(i8 %x) {
+; CHECK-LABEL: 'add_no_nsw_distribute_fold'
+; CHECK-NEXT: Classifying expressions for: @add_no_nsw_distribute_fold
+; CHECK-NEXT: %div = add i8 %x, -2
+; CHECK-NEXT: --> (-2 + %x) U: full-set S: full-set
+; CHECK-NEXT: %div.2 = sdiv i8 %div, -1
+; CHECK-NEXT: --> ((-2 + %x) /s -1) U: [-127,-128) S: [-127,-128)
+; CHECK-NEXT: Determining loop execution counts for: @add_no_nsw_distribute_fold
+;
+ %div = add i8 %x, -2
+ %div.2 = sdiv i8 %div, -1
+ ret i8 %div.2
+}
+
+define i8 @mul_nsw_distribute_fold(i8 range(i8 -16, 0) %x) {
+; CHECK-LABEL: 'mul_nsw_distribute_fold'
+; CHECK-NEXT: Classifying expressions for: @mul_nsw_distribute_fold
+; CHECK-NEXT: %div = mul i8 %x, -2
+; CHECK-NEXT: --> (-2 * %x)<nsw> U: [2,33) S: [2,33)
+; CHECK-NEXT: %div.2 = sdiv i8 %div, -1
+; CHECK-NEXT: --> (2 * %x)<nsw> U: [-32,-1) S: [-32,-1)
+; CHECK-NEXT: Determining loop execution counts for: @mul_nsw_distribute_fold
+;
+ %div = mul i8 %x, -2
+ %div.2 = sdiv i8 %div, -1
+ ret i8 %div.2
+}
+
+define i8 @mul_no_nsw_distribute_fold(i8 %x) {
+; CHECK-LABEL: 'mul_no_nsw_distribute_fold'
+; CHECK-NEXT: Classifying expressions for: @mul_no_nsw_distribute_fold
+; CHECK-NEXT: %div = mul i8 %x, -2
+; CHECK-NEXT: --> (-2 * %x) U: [0,-1) S: [-128,127)
+; CHECK-NEXT: %div.2 = sdiv i8 %div, -1
+; CHECK-NEXT: --> ((-2 * %x) /s -1) U: [-127,-128) S: [-126,-128)
+; CHECK-NEXT: Determining loop execution counts for: @mul_no_nsw_distribute_fold
+;
+ %div = mul i8 %x, -2
+ %div.2 = sdiv i8 %div, -1
+ ret i8 %div.2
+}
+
+define i8 @rndup_idiom(i8 %a) {
+; CHECK-LABEL: 'rndup_idiom'
+; CHECK-NEXT: Classifying expressions for: @rndup_idiom
+; CHECK-NEXT: %m.a = mul i8 -32, %a
+; CHECK-NEXT: --> (-32 * %a) U: [0,-31) S: [-128,97)
+; CHECK-NEXT: %add = add i8 24, %m.a
+; CHECK-NEXT: --> (24 + (-32 * %a))<nuw><nsw> U: [24,-7) S: [-104,121)
+; CHECK-NEXT: %div = sdiv i8 %add, -8
+; CHECK-NEXT: --> ((-9 + (-32 * %a)) /s -8) U: [-15,17) S: [-15,17)
+; CHECK-NEXT: Determining loop execution counts for: @rndup_idiom
+;
+ %m.a = mul i8 -32, %a
+ %add = add i8 24, %m.a
+ %div = sdiv i8 %add, -8
+ ret i8 %div
+}
+
+define i8 @smax_idiom(i8 range(i8 1, 128) %c, i8 %x) {
+; CHECK-LABEL: 'smax_idiom'
+; CHECK-NEXT: Classifying expressions for: @smax_idiom
+; CHECK-NEXT: %smax = call i8 @llvm.smax.i8(i8 63, i8 %x)
+; CHECK-NEXT: --> (63 smax %x) U: [63,-128) S: [63,-128)
+; CHECK-NEXT: %add = add i8 %smax, -63
+; CHECK-NEXT: --> (-63 + (63 smax %x))<nsw> U: [0,65) S: [0,65)
+; CHECK-NEXT: %div = sdiv i8 %add, %x
+; CHECK-NEXT: --> 0 U: [0,1) S: [0,1)
+; CHECK-NEXT: Determining loop execution counts for: @smax_idiom
+;
+ %smax = call i8 @llvm.smax(i8 63, i8 %x)
+ %add = add i8 %smax, -63
+ %div = sdiv i8 %add, %x
+ ret i8 %div
+}
+
define dso_local void @_Z4loopi(i32 %width) local_unnamed_addr #0 {
; CHECK-LABEL: '_Z4loopi'
; CHECK-NEXT: Classifying expressions for: @_Z4loopi
diff --git a/llvm/test/CodeGen/Thumb2/mve-float16regloops.ll b/llvm/test/CodeGen/Thumb2/mve-float16regloops.ll
index a9043476b0549..9ac5513e9bede 100644
--- a/llvm/test/CodeGen/Thumb2/mve-float16regloops.ll
+++ b/llvm/test/CodeGen/Thumb2/mve-float16regloops.ll
@@ -991,149 +991,149 @@ if.end61: ; preds = %if.then59, %while.e
define void @fir(ptr nocapture readonly %S, ptr nocapture readonly %pSrc, ptr nocapture %pDst, i32 %blockSize) {
; CHECK-LABEL: fir:
; CHECK: @ %bb.0: @ %entry
+; CHECK-NEXT: cmp r3, #8
+; CHECK-NEXT: blo.w .LBB16_13
+; CHECK-NEXT: @ %bb.1: @ %if.then
+; CHECK-NEXT: lsrs.w r12, r3, #2
+; CHECK-NEXT: it eq
+; CHECK-NEXT: bxeq lr
+; CHECK-NEXT: .LBB16_2: @ %while.body.lr.ph
; CHECK-NEXT: .save {r4, r5, r6, r7, r8, r9, r10, r11, lr}
; CHECK-NEXT: push.w {r4, r5, r6, r7, r8, r9, r10, r11, lr}
; CHECK-NEXT: .pad #20
; CHECK-NEXT: sub sp, #20
-; CHECK-NEXT: cmp r3, #8
-; CHECK-NEXT: str r1, [sp, #16] @ 4-byte Spill
-; CHECK-NEXT: blo.w .LBB16_12
-; CHECK-NEXT: @ %bb.1: @ %if.then
-; CHECK-NEXT: lsrs.w r12, r3, #2
-; CHECK-NEXT: beq.w .LBB16_12
-; CHECK-NEXT: @ %bb.2: @ %while.body.lr.ph
-; CHECK-NEXT: ldrh r4, [r0]
-; CHECK-NEXT: movs r1, #1
-; CHECK-NEXT: ldrd r5, r3, [r0, #4]
-; CHECK-NEXT: sub.w r0, r4, #8
-; CHECK-NEXT: add.w r7, r0, r0, lsr #29
-; CHECK-NEXT: and r0, r0, #7
-; CHECK-NEXT: asrs r6, r7, #3
-; CHECK-NEXT: cmp r6, #1
-; CHECK-NEXT: it gt
-; CHECK-NEXT: asrgt r1, r7, #3
-; CHECK-NEXT: add.w r7, r5, r4, lsl #1
-; CHECK-NEXT: str r1, [sp] @ 4-byte Spill
-; CHECK-NEXT: subs r1, r7, #2
-; CHECK-NEXT: rsbs r7, r4, #0
-; CHECK-NEXT: str r4, [sp, #8] @ 4-byte Spill
-; CHECK-NEXT: str r7, [sp, #4] @ 4-byte Spill
-; CHECK-NEXT: str r0, [sp, #12] @ 4-byte Spill
+; CHECK-NEXT: ldrh r7, [r0]
+; CHECK-NEXT: ldrd r4, r3, [r0, #4]
+; CHECK-NEXT: add.w r0, r4, r7, lsl #1
+; CHECK-NEXT: subs r0, #2
+; CHECK-NEXT: str r0, [sp, #16] @ 4-byte Spill
+; CHECK-NEXT: rsbs r0, r7, #0
+; CHECK-NEXT: strd r0, r7, [sp] @ 8-byte Folded Spill
+; CHECK-NEXT: sub.w r0, r7, #8
+; CHECK-NEXT: and r7, r0, #7
+; CHECK-NEXT: str r7, [sp, #8] @ 4-byte Spill
+; CHECK-NEXT: add.w r0, r0, r0, lsr #29
+; CHECK-NEXT: asrs r7, r0, #3
; CHECK-NEXT: b .LBB16_6
; CHECK-NEXT: .LBB16_3: @ %while.end.loopexit
; CHECK-NEXT: @ in Loop: Header=BB16_6 Depth=1
-; CHECK-NEXT: ldr r0, [sp, #12] @ 4-byte Reload
-; CHECK-NEXT: add.w r6, r6, r0, lsl #1
+; CHECK-NEXT: ldr r0, [sp, #8] @ 4-byte Reload
+; CHECK-NEXT: add.w r5, r5, r0, lsl #1
; CHECK-NEXT: b .LBB16_5
; CHECK-NEXT: .LBB16_4: @ %for.end
; CHECK-NEXT: @ in Loop: Header=BB16_6 Depth=1
-; CHECK-NEXT: ldr r0, [sp, #12] @ 4-byte Reload
+; CHECK-NEXT: ldr r0, [sp, #8] @ 4-byte Reload
; CHECK-NEXT: wls lr, r0, .LBB16_5
; CHECK-NEXT: b .LBB16_10
; CHECK-NEXT: .LBB16_5: @ %while.end
; CHECK-NEXT: @ in Loop: Header=BB16_6 Depth=1
-; CHECK-NEXT: ldr r0, [sp, #4] @ 4-byte Reload
+; CHECK-NEXT: ldr r0, [sp] @ 4-byte Reload
; CHECK-NEXT: subs.w r12, r12, #1
+; CHECK-NEXT: ldr r1, [sp, #12] @ 4-byte Reload
; CHECK-NEXT: vstrb.8 q0, [r2], #8
-; CHECK-NEXT: add.w r0, r6, r0, lsl #1
-; CHECK-NEXT: add.w r5, r0, #8
+; CHECK-NEXT: add.w r0, r5, r0, lsl #1
+; CHECK-NEXT: add.w r4, r0, #8
; CHECK-NEXT: beq.w .LBB16_12
; CHECK-NEXT: .LBB16_6: @ %while.body
; CHECK-NEXT: @ =>This Loop Header: Depth=1
; CHECK-NEXT: @ Child Loop BB16_8 Depth 2
; CHECK-NEXT: @ Child Loop BB16_11 Depth 2
-; CHECK-NEXT: ldr r0, [sp, #16] @ 4-byte Reload
+; CHECK-NEXT: vldrw.u32 q0, [r1], #8
; CHECK-NEXT: ldrh.w lr, [r3, #14]
-; CHECK-NEXT: vldrw.u32 q0, [r0], #8
-; CHECK-NEXT: ldrh.w r10, [r3, #12]
-; CHECK-NEXT: ldrh r7, [r3, #10]
-; CHECK-NEXT: ldrh r4, [r3, #8]
-; CHECK-NEXT: ldrh r6, [r3, #6]
-; CHECK-NEXT: ldrh.w r9, [r3, #4]
-; CHECK-NEXT: ldrh.w r11, [r3, #2]
-; CHECK-NEXT: ldrh.w r8, [r3]
+; CHECK-NEXT: ldrh r6, [r3, #12]
+; CHECK-NEXT: str r1, [sp, #12] @ 4-byte Spill
+; CHECK-NEXT: ldr r1, [sp, #16] @ 4-byte Reload
+; CHECK-NEXT: ldrh r0, [r3, #10]
+; CHECK-NEXT: ldrh r5, [r3, #8]
+; CHECK-NEXT: ldrh.w r9, [r3, #6]
+; CHECK-NEXT: ldrh.w r8, [r3, #4]
+; CHECK-NEXT: ldrh.w r10, [r3, #2]
+; CHECK-NEXT: ldrh.w r11, [r3]
; CHECK-NEXT: vstrb.8 q0, [r1], #8
-; CHECK-NEXT: vldrw.u32 q0, [r5]
-; CHECK-NEXT: str r0, [sp, #16] @ 4-byte Spill
-; CHECK-NEXT: adds r0, r5, #2
-; CHECK-NEXT: vldrw.u32 q1, [r0]
-; CHECK-NEXT: vmul.f16 q0, q0, r8
-; CHECK-NEXT: adds r0, r5, #6
-; CHECK-NEXT: vfma.f16 q0, q1, r11
-; CHECK-NEXT: vldrw.u32 q1, [r5, #4]
+; CHECK-NEXT: vldrw.u32 q0, [r4]
+; CHECK-NEXT: str r1, [sp, #16] @ 4-byte Spill
+; CHECK-NEXT: adds r1, r4, #2
+; CHECK-NEXT: vldrw.u32 q1, [r1]
+; CHECK-NEXT: vmul.f16 q0, q0, r11
+; CHECK-NEXT: adds r1, r4, #6
+; CHECK-NEXT: vfma.f16 q0, q1, r10
+; CHECK-NEXT: vldrw.u32 q1, [r4, #4]
+; CHECK-NEXT: vfma.f16 q0, q1, r8
+; CHECK-NEXT: vldrw.u32 q1, [r1]
+; CHECK-NEXT: add.w r1, r4, #10
; CHECK-NEXT: vfma.f16 q0, q1, r9
-; CHECK-NEXT: vldrw.u32 q1, [r0]
-; CHECK-NEXT: add.w r0, r5, #10
+; CHECK-NEXT: vldrw.u32 q1, [r4, #8]
+; CHECK-NEXT: vfma.f16 q0, q1, r5
+; CHECK-NEXT: vldrw.u32 q1, [r1]
+; CHECK-NEXT: add.w r5, r4, #16
+; CHECK-NEXT: vfma.f16 q0, q1, r0
+; CHECK-NEXT: vldrw.u32 q1, [r4, #12]
+; CHECK-NEXT: add.w r0, r4, #14
; CHECK-NEXT: vfma.f16 q0, q1, r6
-; CHECK-NEXT: vldrw.u32 q1, [r5, #8]
-; CHECK-NEXT: add.w r6, r5, #16
-; CHECK-NEXT: vfma.f16 q0, q1, r4
; CHECK-NEXT: vldrw.u32 q1, [r0]
-; CHECK-NEXT: add.w r0, r5, #14
-; CHECK-NEXT: vfma.f16 q0, q1, r7
-; CHECK-NEXT: vldrw.u32 q1, [r5, #12]
-; CHECK-NEXT: vfma.f16 q0, q1, r10
-; CHECK-NEXT: vldrw.u32 q1, [r0]
-; CHECK-NEXT: ldr r0, [sp, #8] @ 4-byte Reload
+; CHECK-NEXT: ldr r0, [sp, #4] @ 4-byte Reload
; CHECK-NEXT: vfma.f16 q0, q1, lr
; CHECK-NEXT: cmp r0, #16
; CHECK-NEXT: blo .LBB16_9
; CHECK-NEXT: @ %bb.7: @ %for.body.preheader
; CHECK-NEXT: @ in Loop: Header=BB16_6 Depth=1
-; CHECK-NEXT: ldr r0, [sp] @ 4-byte Reload
-; CHECK-NEXT: add.w r5, r3, #16
-; CHECK-NEXT: dls lr, r0
+; CHECK-NEXT: add.w r4, r3, #16
+; CHECK-NEXT: movs r6, #0
; CHECK-NEXT: .LBB16_8: @ %for.body
; CHECK-NEXT: @ Parent Loop BB16_6 Depth=1
; CHECK-NEXT: @ => This Inner Loop Header: Depth=2
-; CHECK-NEXT: ldrh r0, [r5], #16
-; CHECK-NEXT: vldrw.u32 q1, [r6]
-; CHECK-NEXT: adds r4, r6, #2
+; CHECK-NEXT: ldrh r0, [r4], #16
+; CHECK-NEXT: vldrw.u32 q1, [r5]
+; CHECK-NEXT: adds r1, r5, #2
+; CHECK-NEXT: adds r6, #1
; CHECK-NEXT: vfma.f16 q0, q1, r0
-; CHECK-NEXT: vldrw.u32 q1, [r4]
-; CHECK-NEXT: ldrh r0, [r5, #-14]
-; CHECK-NEXT: adds r4, r6, #6
+; CHECK-NEXT: vldrw.u32 q1, [r1]
+; CHECK-NEXT: ldrh r0, [r4, #-14]
+; CHECK-NEXT: adds r1, r5, #6
+; CHECK-NEXT: cmp r6, r7
; CHECK-NEXT: vfma.f16 q0, q1, r0
-; CHECK-NEXT: ldrh r0, [r5, #-12]
-; CHECK-NEXT: vldrw.u32 q1, [r6, #4]
+; CHECK-NEXT: ldrh r0, [r4, #-12]
+; CHECK-NEXT: vldrw.u32 q1, [r5, #4]
; CHECK-NEXT: vfma.f16 q0, q1, r0
-; CHECK-NEXT: vldrw.u32 q1, [r4]
-; CHECK-NEXT: ldrh r0, [r5, #-10]
-; CHECK-NEXT: add.w r4, r6, #10
+; CHECK-NEXT: vldrw.u32 q1, [r1]
+; CHECK-NEXT: ldrh r0, [r4, #-10]
+; CHECK-NEXT: add.w r1, r5, #10
; CHECK-NEXT: vfma.f16 q0, q1, r0
-; CHECK-NEXT: ldrh r0, [r5, #-8]
-; CHECK-NEXT: vldrw.u32 q1, [r6, #8]
+; CHECK-NEXT: ldrh r0, [r4, #-8]
+; CHECK-NEXT: vldrw.u32 q1, [r5, #8]
; CHECK-NEXT: vfma.f16 q0, q1, r0
-; CHECK-NEXT: vldrw.u32 q1, [r4]
-; CHECK-NEXT: ldrh r0, [r5, #-6]
-; CHECK-NEXT: ldrh r4, [r5, #-2]
+; CHECK-NEXT: vldrw.u32 q1, [r1]
+; CHECK-NEXT: ldrh r0, [r4, #-6]
+; CHECK-NEXT: ldrh r1, [r4, #-2]
; CHECK-NEXT: vfma.f16 q0, q1, r0
-; CHECK-NEXT: ldrh r0, [r5, #-4]
-; CHECK-NEXT: vldrw.u32 q1, [r6, #12]
+; CHECK-NEXT: ldrh r0, [r4, #-4]
+; CHECK-NEXT: vldrw.u32 q1, [r5, #12]
; CHECK-NEXT: vfma.f16 q0, q1, r0
-; CHECK-NEXT: add.w r0, r6, #14
+; CHECK-NEXT: add.w r0, r5, #14
; CHECK-NEXT: vldrw.u32 q1, [r0]
-; CHECK-NEXT: adds r6, #16
-; CHECK-NEXT: vfma.f16 q0, q1, r4
-; CHECK-NEXT: le lr, .LBB16_8
+; CHECK-NEXT: add.w r5, r5, #16
+; CHECK-NEXT: vfma.f16 q0, q1, r1
+; CHECK-NEXT: blt .LBB16_8
; CHECK-NEXT: b .LBB16_4
; CHECK-NEXT: .LBB16_9: @ in Loop: Header=BB16_6 Depth=1
-; CHECK-NEXT: add.w r5, r3, #16
+; CHECK-NEXT: add.w r4, r3, #16
; CHECK-NEXT: b .LBB16_4
; CHECK-NEXT: .LBB16_10: @ %while.body76.preheader
; CHECK-NEXT: @ in Loop: Header=BB16_6 Depth=1
-; CHECK-NEXT: mov r0, r6
+; CHECK-NEXT: mov r0, r5
; CHECK-NEXT: .LBB16_11: @ %while.body76
; CHECK-NEXT: @ Parent Loop BB16_6 Depth=1
; CHECK-NEXT: @ => This Inner Loop Header: Depth=2
-; CHECK-NEXT: ldrh r4, [r5], #2
+; CHECK-NEXT: ldrh r1, [r4], #2
; CHECK-NEXT: vldrh.u16 q1, [r0], #2
-; CHECK-NEXT: vfma.f16 q0, q1, r4
+; CHECK-NEXT: vfma.f16 q0, q1, r1
; CHECK-NEXT: le lr, .LBB16_11
; CHECK-NEXT: b .LBB16_3
-; CHECK-NEXT: .LBB16_12: @ %if.end
+; CHECK-NEXT: .LBB16_12:
; CHECK-NEXT: add sp, #20
-; CHECK-NEXT: pop.w {r4, r5, r6, r7, r8, r9, r10, r11, pc}
+; CHECK-NEXT: pop.w {r4, r5, r6, r7, r8, r9, r10, r11, lr}
+; CHECK-NEXT: .LBB16_13: @ %if.end
+; CHECK-NEXT: bx lr
entry:
%pState1 = getelementptr inbounds %struct.arm_fir_instance_f32, ptr %S, i32 0, i32 1
%i = load ptr, ptr %pState1, align 4
diff --git a/llvm/test/CodeGen/Thumb2/mve-float32regloops.ll b/llvm/test/CodeGen/Thumb2/mve-float32regloops.ll
index b6657d607ce6d..296b263420c23 100644
--- a/llvm/test/CodeGen/Thumb2/mve-float32regloops.ll
+++ b/llvm/test/CodeGen/Thumb2/mve-float32regloops.ll
@@ -983,12 +983,9 @@ define void @fir(ptr nocapture readonly %S, ptr nocapture readonly %pSrc, ptr no
; CHECK-LABEL: fir:
; CHECK: @ %bb.0: @ %entry
; CHECK-NEXT: cmp r3, #8
-; CHECK-NEXT: blo.w .LBB16_13
-; CHECK-NEXT: @ %bb.1: @ %if.then
-; CHECK-NEXT: lsrs.w r12, r3, #2
-; CHECK-NEXT: it eq
-; CHECK-NEXT: bxeq lr
-; CHECK-NEXT: .LBB16_2: @ %while.body.lr.ph
+; CHECK-NEXT: it lo
+; CHECK-NEXT: bxlo lr
+; CHECK-NEXT: .LBB16_1: @ %if.then
; CHECK-NEXT: .save {r4, r5, r6, r7, r8, r9, r10, r11, lr}
; CHECK-NEXT: push.w {r4, r5, r6, r7, r8, r9, r10, r11, lr}
; CHECK-NEXT: .pad #4
@@ -997,123 +994,132 @@ define void @fir(ptr nocapture readonly %S, ptr nocapture readonly %pSrc, ptr no
; CHECK-NEXT: vpush {d8, d9, d10, d11, d12, d13}
; CHECK-NEXT: .pad #24
; CHECK-NEXT: sub sp, #24
-; CHECK-NEXT: ldrh r6, [r0]
-; CHECK-NEXT: movs r4, #1
-; CHECK-NEXT: ldrd r7, r10, [r0, #4]
-; CHECK-NEXT: sub.w r0, r6, #8
-; CHECK-NEXT: add.w r3, r0, r0, lsr #29
-; CHECK-NEXT: and r0, r0, #7
-; CHECK-NEXT: asrs r5, r3, #3
-; CHECK-NEXT: cmp r5, #1
-; CHECK-NEXT: it gt
-; CHECK-NEXT: asrgt r4, r3, #3
-; CHECK-NEXT: add.w r3, r7, r6, lsl #2
-; CHECK-NEXT: sub.w r9, r3, #4
-; CHECK-NEXT: rsbs r3, r6, #0
-; CHECK-NEXT: str r4, [sp] @ 4-byte Spill
-; CHECK-NEXT: str r6, [sp, #8] @ 4-byte Spill
-; CHECK-NEXT: str r3, [sp, #4] @ 4-byte Spill
-; CHECK-NEXT: str r0, [sp, #12] @ 4-byte Spill
+; CHECK-NEXT: lsrs r6, r3, #2
+; CHECK-NEXT: beq.w .LBB16_13
+; CHECK-NEXT: @ %bb.2: @ %while.body.lr.ph
+; CHECK-NEXT: ldrh r7, [r0]
+; CHECK-NEXT: mov r8, r1
+; CHECK-NEXT: ldrd r3, r9, [r0, #4]
+; CHECK-NEXT: add.w r0, r3, r7, lsl #2
+; CHECK-NEXT: subs r0, #4
+; CHECK-NEXT: str r0, [sp, #20] @ 4-byte Spill
+; CHECK-NEXT: rsbs r0, r7, #0
+; CHECK-NEXT: strd r0, r7, [sp, #4] @ 8-byte Folded Spill
+; CHECK-NEXT: sub.w r0, r7, #8
+; CHECK-NEXT: and r1, r0, #7
+; CHECK-NEXT: str r1, [sp, #12] @ 4-byte Spill
+; CHECK-NEXT: add.w r0, r0, r0, lsr #29
+; CHECK-NEXT: asr.w r12, r0, #3
; CHECK-NEXT: b .LBB16_6
; CHECK-NEXT: .LBB16_3: @ %while.end.loopexit
; CHECK-NEXT: @ in Loop: Header=BB16_6 Depth=1
; CHECK-NEXT: ldr r0, [sp, #12] @ 4-byte Reload
-; CHECK-NEXT: add.w r7, r7, r0, lsl #2
+; CHECK-NEXT: add.w r3, r3, r0, lsl #2
; CHECK-NEXT: b .LBB16_5
; CHECK-NEXT: .LBB16_4: @ %for.end
; CHECK-NEXT: @ in Loop: Header=BB16_6 Depth=1
-; CHECK-NEXT: ldr r1, [sp, #20] @ 4-byte Reload
-; CHECK-NEXT: ldrd r0, r9, [sp, #12] @ 8-byte Folded Reload
+; CHECK-NEXT: ldr r0, [sp, #12] @ 4-byte Reload
; CHECK-NEXT: wls lr, r0, .LBB16_5
-; CHECK-NEXT: b .LBB16_10
+; CHECK-NEXT: b .LBB16_11
; CHECK-NEXT: .LBB16_5: @ %while.end
; CHECK-NEXT: @ in Loop: Header=BB16_6 Depth=1
; CHECK-NEXT: ldr r0, [sp, #4] @ 4-byte Reload
-; CHECK-NEXT: subs.w r12, r12, #1
+; CHECK-NEXT: subs r6, #1
; CHECK-NEXT: vstrb.8 q0, [r2], #16
-; CHECK-NEXT: add.w r0, r7, r0, lsl #2
-; CHECK-NEXT: add.w r7, r0, #16
-; CHECK-NEXT: beq .LBB16_12
+; CHECK-NEXT: add.w r0, r3, r0, lsl #2
+; CHECK-NEXT: add.w r3, r0, #16
+; CHECK-NEXT: beq .LBB16_13
; CHECK-NEXT: .LBB16_6: @ %while.body
; CHECK-NEXT: @ =>This Loop Header: Depth=1
; CHECK-NEXT: @ Child Loop BB16_8 Depth 2
-; CHECK-NEXT: @ Child Loop BB16_11 Depth 2
-; CHECK-NEXT: add.w lr, r10, #8
-; CHECK-NEXT: vldrw.u32 q0, [r1], #16
-; CHECK-NEXT: ldrd r3, r4, [r10]
-; CHECK-NEXT: ldm.w lr, {r0, r5, r6, lr}
-; CHECK-NEXT: ldrd r11, r8, [r10, #24]
-; CHECK-NEXT: vstrb.8 q0, [r9], #16
-; CHECK-NEXT: vldrw.u32 q0, [r7], #32
-; CHECK-NEXT: str r1, [sp, #20] @ 4-byte Spill
-; CHECK-NEXT: str.w r9, [sp, #16] @ 4-byte Spill
-; CHECK-NEXT: vldrw.u32 q1, [r7, #-28]
-; CHECK-NEXT: vmul.f32 q0, q0, r3
-; CHECK-NEXT: vldrw.u32 q6, [r7, #-24]
-; CHECK-NEXT: vldrw.u32 q4, [r7, #-20]
-; CHECK-NEXT: vfma.f32 q0, q1, r4
-; CHECK-NEXT: vldrw.u32 q5, [r7, #-16]
-; CHECK-NEXT: vfma.f32 q0, q6, r0
-; CHECK-NEXT: vldrw.u32 q2, [r7, #-12]
-; CHECK-NEXT: vfma.f32 q0, q4, r5
-; CHECK-NEXT: vldrw.u32 q3, [r7, #-8]
-; CHECK-NEXT: vfma.f32 q0, q5, r6
-; CHECK-NEXT: vldrw.u32 q1, [r7, #-4]
-; CHECK-NEXT: vfma.f32 q0, q2, lr
+; CHECK-NEXT: @ Child Loop BB16_12 Depth 2
+; CHECK-NEXT: vldrw.u32 q0, [r8], #16
+; CHECK-NEXT: add.w lr, r9, #8
+; CHECK-NEXT: ldrd r0, r5, [r9]
+; CHECK-NEXT: str.w r8, [sp, #16] @ 4-byte Spill
+; CHECK-NEXT: ldr.w r8, [sp, #20] @ 4-byte Reload
+; CHECK-NEXT: ldm.w lr, {r4, r7, lr}
+; CHECK-NEXT: ldrd r10, r11, [r9, #20]
+; CHECK-NEXT: ldr.w r1, [r9, #28]
+; CHECK-NEXT: vstrb.8 q0, [r8], #16
+; CHECK-NEXT: vldrw.u32 q0, [r3], #32
+; CHECK-NEXT: str.w r8, [sp, #20] @ 4-byte Spill
+; CHECK-NEXT: vldrw.u32 q1, [r3, #-28]
+; CHECK-NEXT: vmul.f32 q0, q0, r0
+; CHECK-NEXT: vldrw.u32 q6, [r3, #-24]
+; CHECK-NEXT: vldrw.u32 q4, [r3, #-20]
+; CHECK-NEXT: vfma.f32 q0, q1, r5
+; CHECK-NEXT: vldrw.u32 q5, [r3, #-16]
+; CHECK-NEXT: vfma.f32 q0, q6, r4
+; CHECK-NEXT: vldrw.u32 q2, [r3, #-12]
+; CHECK-NEXT: vfma.f32 q0, q4, r7
+; CHECK-NEXT: vldrw.u32 q3, [r3, #-8]
+; CHECK-NEXT: vfma.f32 q0, q5, lr
+; CHECK-NEXT: vldrw.u32 q1, [r3, #-4]
+; CHECK-NEXT: vfma.f32 q0, q2, r10
; CHECK-NEXT: ldr r0, [sp, #8] @ 4-byte Reload
; CHECK-NEXT: vfma.f32 q0, q3, r11
-; CHECK-NEXT: vfma.f32 q0, q1, r8
+; CHECK-NEXT: vfma.f32 q0, q1, r1
; CHECK-NEXT: cmp r0, #16
-; CHECK-NEXT: blo .LBB16_9
+; CHECK-NEXT: blo .LBB16_10
; CHECK-NEXT: @ %bb.7: @ %for.body.preheader
; CHECK-NEXT: @ in Loop: Header=BB16_6 Depth=1
-; CHECK-NEXT: ldr r0, [sp] @ 4-byte Reload
-; CHECK-NEXT: add.w r4, r10, #32
-; CHECK-NEXT: dls lr, r0
+; CHECK-NEXT: add.w r5, r9, #32
+; CHECK-NEXT: mov r11, r2
+; CHECK-NEXT: movs r0, #0
+; CHECK-NEXT: str r6, [sp] @ 4-byte Spill
; CHECK-NEXT: .LBB16_8: @ %for.body
; CHECK-NEXT: @ Parent Loop BB16_6 Depth=1
; CHECK-NEXT: @ => This Inner Loop Header: Depth=2
-; CHECK-NEXT: ldm.w r4, {r0, r3, r5, r6, r8, r11}
-; CHECK-NEXT: vldrw.u32 q1, [r7], #32
-; CHECK-NEXT: vldrw.u32 q6, [r7, #-24]
-; CHECK-NEXT: vldrw.u32 q4, [r7, #-20]
-; CHECK-NEXT: vfma.f32 q0, q1, r0
-; CHECK-NEXT: vldrw.u32 q1, [r7, #-28]
-; CHECK-NEXT: vldrw.u32 q5, [r7, #-16]
-; CHECK-NEXT: vldrw.u32 q2, [r7, #-12]
-; CHECK-NEXT: vfma.f32 q0, q1, r3
-; CHECK-NEXT: ldrd r9, r1, [r4, #24]
-; CHECK-NEXT: vfma.f32 q0, q6, r5
-; CHECK-NEXT: vldrw.u32 q3, [r7, #-8]
+; CHECK-NEXT: vldrw.u32 q1, [r3], #32
+; CHECK-NEXT: ldrd r7, r4, [r5]
+; CHECK-NEXT: ldrd r1, r6, [r5, #8]
+; CHECK-NEXT: adds r0, #1
+; CHECK-NEXT: vfma.f32 q0, q1, r7
+; CHECK-NEXT: vldrw.u32 q1, [r3, #-28]
+; CHECK-NEXT: vldrw.u32 q6, [r3, #-24]
+; CHECK-NEXT: vldrw.u32 q4, [r3, #-20]
+; CHECK-NEXT: vfma.f32 q0, q1, r4
+; CHECK-NEXT: ldrd r2, lr, [r5, #16]
+; CHECK-NEXT: vfma.f32 q0, q6, r1
+; CHECK-NEXT: vldrw.u32 q5, [r3, #-16]
; CHECK-NEXT: vfma.f32 q0, q4, r6
-; CHECK-NEXT: vldrw.u32 q1, [r7, #-4]
-; CHECK-NEXT: vfma.f32 q0, q5, r8
-; CHECK-NEXT: adds r4, #32
-; CHECK-NEXT: vfma.f32 q0, q2, r11
-; CHECK-NEXT: vfma.f32 q0, q3, r9
-; CHECK-NEXT: vfma.f32 q0, q1, r1
-; CHECK-NEXT: le lr, .LBB16_8
+; CHECK-NEXT: vldrw.u32 q2, [r3, #-12]
+; CHECK-NEXT: vfma.f32 q0, q5, r2
+; CHECK-NEXT: ldrd r8, r10, [r5, #24]
+; CHECK-NEXT: vldrw.u32 q3, [r3, #-8]
+; CHECK-NEXT: vfma.f32 q0, q2, lr
+; CHECK-NEXT: vldrw.u32 q1, [r3, #-4]
+; CHECK-NEXT: adds r5, #32
+; CHECK-NEXT: vfma.f32 q0, q3, r8
+; CHECK-NEXT: cmp r0, r12
+; CHECK-NEXT: vfma.f32 q0, q1, r10
+; CHECK-NEXT: blt .LBB16_8
+; CHECK-NEXT: @ %bb.9: @ in Loop: Header=BB16_6 Depth=1
+; CHECK-NEXT: ldr.w r8, [sp, #16] @ 4-byte Reload
+; CHECK-NEXT: mov r2, r11
+; CHECK-NEXT: ldr r6, [sp] @ 4-byte Reload
; CHECK-NEXT: b .LBB16_4
-; CHECK-NEXT: .LBB16_9: @ in Loop: Header=BB16_6 Depth=1
-; CHECK-NEXT: add.w r4, r10, #32
+; CHECK-NEXT: .LBB16_10: @ in Loop: Header=BB16_6 Depth=1
+; CHECK-NEXT: add.w r5, r9, #32
+; CHECK-NEXT: ldr.w r8, [sp, #16] @ 4-byte Reload
; CHECK-NEXT: b .LBB16_4
-; CHECK-NEXT: .LBB16_10: @ %while.body76.preheader
+; CHECK-NEXT: .LBB16_11: @ %while.body76.preheader
; CHECK-NEXT: @ in Loop: Header=BB16_6 Depth=1
-; CHECK-NEXT: mov r3, r7
-; CHECK-NEXT: .LBB16_11: @ %while.body76
+; CHECK-NEXT: mov r0, r3
+; CHECK-NEXT: .LBB16_12: @ %while.body76
; CHECK-NEXT: @ Parent Loop BB16_6 Depth=1
; CHECK-NEXT: @ => This Inner Loop Header: Depth=2
-; CHECK-NEXT: ldr r0, [r4], #4
-; CHECK-NEXT: vldrw.u32 q1, [r3], #4
-; CHECK-NEXT: vfma.f32 q0, q1, r0
-; CHECK-NEXT: le lr, .LBB16_11
+; CHECK-NEXT: ldr r4, [r5], #4
+; CHECK-NEXT: vldrw.u32 q1, [r0], #4
+; CHECK-NEXT: vfma.f32 q0, q1, r4
+; CHECK-NEXT: le lr, .LBB16_12
; CHECK-NEXT: b .LBB16_3
-; CHECK-NEXT: .LBB16_12:
+; CHECK-NEXT: .LBB16_13:
; CHECK-NEXT: add sp, #24
; CHECK-NEXT: vpop {d8, d9, d10, d11, d12, d13}
; CHECK-NEXT: add sp, #4
; CHECK-NEXT: pop.w {r4, r5, r6, r7, r8, r9, r10, r11, lr}
-; CHECK-NEXT: .LBB16_13: @ %if.end
; CHECK-NEXT: bx lr
entry:
%pState1 = getelementptr inbounds %struct.arm_fir_instance_f32, ptr %S, i32 0, i32 1
diff --git a/llvm/test/CodeGen/X86/optimize-max-0.ll b/llvm/test/CodeGen/X86/optimize-max-0.ll
index b6af7e1641a9c..e89e4d74212a1 100644
--- a/llvm/test/CodeGen/X86/optimize-max-0.ll
+++ b/llvm/test/CodeGen/X86/optimize-max-0.ll
@@ -14,198 +14,211 @@ define void @foo(ptr %r, i32 %s, i32 %w, i32 %x, ptr %j, i32 %d) nounwind {
; CHECK-NEXT: pushl %ebx
; CHECK-NEXT: pushl %edi
; CHECK-NEXT: pushl %esi
-; CHECK-NEXT: subl $28, %esp
-; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edi
-; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edx
-; CHECK-NEXT: movl {{[0-9]+}}(%esp), %esi
+; CHECK-NEXT: subl $44, %esp
; CHECK-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; CHECK-NEXT: movl %edx, %eax
-; CHECK-NEXT: imull %esi, %eax
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edi
+; CHECK-NEXT: movl %ebx, %eax
+; CHECK-NEXT: imull %ebp, %eax
+; CHECK-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
; CHECK-NEXT: cmpl $1, {{[0-9]+}}(%esp)
-; CHECK-NEXT: movl %eax, (%esp) ## 4-byte Spill
-; CHECK-NEXT: je LBB0_19
+; CHECK-NEXT: je LBB0_20
; CHECK-NEXT: ## %bb.1: ## %bb10.preheader
-; CHECK-NEXT: movl %eax, %ebp
-; CHECK-NEXT: sarl $31, %ebp
-; CHECK-NEXT: shrl $30, %ebp
-; CHECK-NEXT: addl %eax, %ebp
-; CHECK-NEXT: sarl $2, %ebp
-; CHECK-NEXT: testl %edx, %edx
+; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax ## 4-byte Reload
+; CHECK-NEXT: movl %eax, %ecx
+; CHECK-NEXT: sarl $31, %ecx
+; CHECK-NEXT: shrl $30, %ecx
+; CHECK-NEXT: addl %eax, %ecx
+; CHECK-NEXT: sarl $2, %ecx
+; CHECK-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
+; CHECK-NEXT: testl %ebx, %ebx
; CHECK-NEXT: jle LBB0_12
; CHECK-NEXT: ## %bb.2: ## %bb.nph9
-; CHECK-NEXT: testl %esi, %esi
+; CHECK-NEXT: testl %ebp, %ebp
; CHECK-NEXT: jle LBB0_12
; CHECK-NEXT: ## %bb.3: ## %bb.nph9.split
-; CHECK-NEXT: movl {{[0-9]+}}(%esp), %eax
-; CHECK-NEXT: incl %eax
+; CHECK-NEXT: leal 1(%edi), %eax
; CHECK-NEXT: xorl %ecx, %ecx
-; CHECK-NEXT: movl %edi, %edx
-; CHECK-NEXT: xorl %edi, %edi
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edx
+; CHECK-NEXT: xorl %esi, %esi
; CHECK-NEXT: .p2align 4
; CHECK-NEXT: LBB0_4: ## %bb6
; CHECK-NEXT: ## =>This Inner Loop Header: Depth=1
-; CHECK-NEXT: movzbl (%eax,%edi,2), %ebx
-; CHECK-NEXT: movb %bl, (%edx,%edi)
-; CHECK-NEXT: incl %edi
-; CHECK-NEXT: cmpl %esi, %edi
+; CHECK-NEXT: movzbl (%eax,%esi,2), %ebx
+; CHECK-NEXT: movb %bl, (%edx,%esi)
+; CHECK-NEXT: incl %esi
+; CHECK-NEXT: cmpl %ebp, %esi
; CHECK-NEXT: jl LBB0_4
; CHECK-NEXT: ## %bb.5: ## %bb9
; CHECK-NEXT: ## in Loop: Header=BB0_4 Depth=1
; CHECK-NEXT: incl %ecx
; CHECK-NEXT: addl {{[0-9]+}}(%esp), %eax
-; CHECK-NEXT: addl %esi, %edx
-; CHECK-NEXT: cmpl {{[0-9]+}}(%esp), %ecx
+; CHECK-NEXT: addl %ebp, %edx
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %ebx
+; CHECK-NEXT: cmpl %ebx, %ecx
; CHECK-NEXT: je LBB0_12
; CHECK-NEXT: ## %bb.6: ## %bb7.preheader
; CHECK-NEXT: ## in Loop: Header=BB0_4 Depth=1
-; CHECK-NEXT: xorl %edi, %edi
+; CHECK-NEXT: xorl %esi, %esi
; CHECK-NEXT: jmp LBB0_4
; CHECK-NEXT: LBB0_12: ## %bb18.loopexit
-; CHECK-NEXT: movl %ebp, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
-; CHECK-NEXT: movl (%esp), %eax ## 4-byte Reload
-; CHECK-NEXT: addl %ebp, %eax
+; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax ## 4-byte Reload
+; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx ## 4-byte Reload
+; CHECK-NEXT: addl %ecx, %eax
; CHECK-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
-; CHECK-NEXT: cmpl $1, {{[0-9]+}}(%esp)
+; CHECK-NEXT: cmpl $1, %ebx
; CHECK-NEXT: jle LBB0_13
; CHECK-NEXT: ## %bb.7: ## %bb.nph5
-; CHECK-NEXT: cmpl $2, %esi
+; CHECK-NEXT: cmpl $2, {{[0-9]+}}(%esp)
; CHECK-NEXT: jl LBB0_13
; CHECK-NEXT: ## %bb.8: ## %bb.nph5.split
-; CHECK-NEXT: movl %esi, %ebp
-; CHECK-NEXT: shrl $31, %ebp
-; CHECK-NEXT: addl %esi, %ebp
-; CHECK-NEXT: sarl %ebp
; CHECK-NEXT: movl {{[0-9]+}}(%esp), %eax
; CHECK-NEXT: movl %eax, %ecx
; CHECK-NEXT: shrl $31, %ecx
; CHECK-NEXT: addl %eax, %ecx
; CHECK-NEXT: sarl %ecx
-; CHECK-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
-; CHECK-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax ## 4-byte Reload
-; CHECK-NEXT: addl %ecx, %eax
-; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edx
-; CHECK-NEXT: addl $2, %edx
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %eax
+; CHECK-NEXT: movl %eax, %edx
+; CHECK-NEXT: shrl $31, %edx
+; CHECK-NEXT: addl %eax, %edx
+; CHECK-NEXT: sarl %edx
; CHECK-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
-; CHECK-NEXT: movl (%esp), %edx ## 4-byte Reload
-; CHECK-NEXT: addl %edx, %ecx
-; CHECK-NEXT: xorl %edi, %edi
-; CHECK-NEXT: xorl %edx, %edx
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %eax
+; CHECK-NEXT: addl $2, %eax
+; CHECK-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
+; CHECK-NEXT: xorl %eax, %eax
+; CHECK-NEXT: xorl %esi, %esi
; CHECK-NEXT: .p2align 4
; CHECK-NEXT: LBB0_9: ## %bb13
; CHECK-NEXT: ## =>This Loop Header: Depth=1
; CHECK-NEXT: ## Child Loop BB0_10 Depth 2
-; CHECK-NEXT: movl %edi, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
-; CHECK-NEXT: andl $1, %edi
-; CHECK-NEXT: movl %edx, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
-; CHECK-NEXT: addl %edx, %edi
-; CHECK-NEXT: imull {{[0-9]+}}(%esp), %edi
-; CHECK-NEXT: addl {{[-0-9]+}}(%e{{[sb]}}p), %edi ## 4-byte Folded Reload
+; CHECK-NEXT: movl %eax, %edx
+; CHECK-NEXT: andl $1, %edx
+; CHECK-NEXT: movl %esi, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
+; CHECK-NEXT: addl %esi, %edx
+; CHECK-NEXT: imull {{[0-9]+}}(%esp), %edx
+; CHECK-NEXT: addl {{[-0-9]+}}(%e{{[sb]}}p), %edx ## 4-byte Folded Reload
+; CHECK-NEXT: movl %ecx, %esi
+; CHECK-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
+; CHECK-NEXT: imull %eax, %esi
+; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax ## 4-byte Reload
+; CHECK-NEXT: addl %esi, %eax
+; CHECK-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
+; CHECK-NEXT: addl {{[-0-9]+}}(%e{{[sb]}}p), %esi ## 4-byte Folded Reload
; CHECK-NEXT: xorl %ebx, %ebx
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %ebp
; CHECK-NEXT: .p2align 4
; CHECK-NEXT: LBB0_10: ## %bb14
; CHECK-NEXT: ## Parent Loop BB0_9 Depth=1
; CHECK-NEXT: ## => This Inner Loop Header: Depth=2
-; CHECK-NEXT: movzbl -2(%edi,%ebx,4), %edx
-; CHECK-NEXT: movb %dl, (%ecx,%ebx)
-; CHECK-NEXT: movzbl (%edi,%ebx,4), %edx
-; CHECK-NEXT: movb %dl, (%eax,%ebx)
+; CHECK-NEXT: movl %ecx, %eax
+; CHECK-NEXT: movzbl -2(%edx,%ebx,4), %ecx
+; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edi ## 4-byte Reload
+; CHECK-NEXT: addl %ebx, %edi
+; CHECK-NEXT: movb %cl, (%ebp,%edi)
+; CHECK-NEXT: movzbl (%edx,%ebx,4), %ecx
+; CHECK-NEXT: leal (%esi,%ebx), %edi
+; CHECK-NEXT: movb %cl, (%ebp,%edi)
+; CHECK-NEXT: movl %eax, %ecx
; CHECK-NEXT: incl %ebx
-; CHECK-NEXT: cmpl %ebp, %ebx
+; CHECK-NEXT: cmpl %eax, %ebx
; CHECK-NEXT: jl LBB0_10
; CHECK-NEXT: ## %bb.11: ## %bb17
; CHECK-NEXT: ## in Loop: Header=BB0_9 Depth=1
-; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edi ## 4-byte Reload
-; CHECK-NEXT: incl %edi
-; CHECK-NEXT: addl %ebp, %eax
-; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx ## 4-byte Reload
-; CHECK-NEXT: addl $2, %edx
-; CHECK-NEXT: addl %ebp, %ecx
-; CHECK-NEXT: cmpl {{[-0-9]+}}(%e{{[sb]}}p), %edi ## 4-byte Folded Reload
+; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %eax ## 4-byte Reload
+; CHECK-NEXT: incl %eax
+; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %esi ## 4-byte Reload
+; CHECK-NEXT: addl $2, %esi
+; CHECK-NEXT: cmpl {{[-0-9]+}}(%e{{[sb]}}p), %eax ## 4-byte Folded Reload
; CHECK-NEXT: jl LBB0_9
; CHECK-NEXT: LBB0_13: ## %bb20
-; CHECK-NEXT: movl {{[0-9]+}}(%esp), %ecx
-; CHECK-NEXT: cmpl $1, %ecx
-; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edx
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %eax
+; CHECK-NEXT: cmpl $1, %eax
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %esi
; CHECK-NEXT: movl {{[0-9]+}}(%esp), %ebx
-; CHECK-NEXT: je LBB0_19
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %ebp
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edi
+; CHECK-NEXT: je LBB0_20
; CHECK-NEXT: ## %bb.14: ## %bb20
-; CHECK-NEXT: cmpl $3, %ecx
+; CHECK-NEXT: cmpl $3, %eax
; CHECK-NEXT: jne LBB0_24
; CHECK-NEXT: ## %bb.15: ## %bb22
-; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ebp ## 4-byte Reload
-; CHECK-NEXT: addl %ebp, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Folded Spill
-; CHECK-NEXT: testl %edx, %edx
+; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edx ## 4-byte Reload
+; CHECK-NEXT: addl {{[-0-9]+}}(%e{{[sb]}}p), %edx ## 4-byte Folded Reload
+; CHECK-NEXT: testl %ebx, %ebx
; CHECK-NEXT: jle LBB0_18
; CHECK-NEXT: ## %bb.16: ## %bb.nph
-; CHECK-NEXT: leal 15(%edx), %eax
+; CHECK-NEXT: leal 15(%ebx), %eax
; CHECK-NEXT: andl $-16, %eax
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; CHECK-NEXT: addl $15, %ecx
+; CHECK-NEXT: andl $-16, %ecx
+; CHECK-NEXT: movl %ecx, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
; CHECK-NEXT: imull {{[0-9]+}}(%esp), %eax
-; CHECK-NEXT: addl %ebp, %ebp
-; CHECK-NEXT: movl (%esp), %ecx ## 4-byte Reload
-; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edi
-; CHECK-NEXT: addl %edi, %ecx
-; CHECK-NEXT: addl %ecx, %ebp
-; CHECK-NEXT: addl %eax, %ebx
-; CHECK-NEXT: leal 15(%esi), %eax
-; CHECK-NEXT: andl $-16, %eax
-; CHECK-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
+; CHECK-NEXT: addl %eax, %edi
+; CHECK-NEXT: xorl %ebp, %ebp
; CHECK-NEXT: .p2align 4
; CHECK-NEXT: LBB0_17: ## %bb23
; CHECK-NEXT: ## =>This Inner Loop Header: Depth=1
+; CHECK-NEXT: movl %edi, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
+; CHECK-NEXT: movl %ebp, %eax
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %ecx
+; CHECK-NEXT: imull %ecx, %eax
+; CHECK-NEXT: addl %edx, %eax
+; CHECK-NEXT: addl %esi, %eax
; CHECK-NEXT: subl $4, %esp
-; CHECK-NEXT: pushl %esi
-; CHECK-NEXT: pushl %ebx
-; CHECK-NEXT: pushl %ebp
-; CHECK-NEXT: movl %ebp, %edi
-; CHECK-NEXT: movl %ebx, %ebp
-; CHECK-NEXT: movl %edx, %ebx
+; CHECK-NEXT: pushl %ecx
+; CHECK-NEXT: pushl %edi
+; CHECK-NEXT: pushl %eax
+; CHECK-NEXT: movl %esi, %edi
+; CHECK-NEXT: movl %edx, %esi
; CHECK-NEXT: calll _memcpy
-; CHECK-NEXT: movl %ebx, %edx
-; CHECK-NEXT: movl %ebp, %ebx
-; CHECK-NEXT: movl %edi, %ebp
+; CHECK-NEXT: movl %esi, %edx
+; CHECK-NEXT: movl %edi, %esi
+; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %edi ## 4-byte Reload
; CHECK-NEXT: addl $16, %esp
-; CHECK-NEXT: addl %esi, %ebp
-; CHECK-NEXT: addl {{[-0-9]+}}(%e{{[sb]}}p), %ebx ## 4-byte Folded Reload
-; CHECK-NEXT: decl %edx
+; CHECK-NEXT: incl %ebp
+; CHECK-NEXT: addl {{[-0-9]+}}(%e{{[sb]}}p), %edi ## 4-byte Folded Reload
+; CHECK-NEXT: decl %ebx
; CHECK-NEXT: jne LBB0_17
; CHECK-NEXT: LBB0_18: ## %bb26
-; CHECK-NEXT: movl (%esp), %ecx ## 4-byte Reload
-; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %esi ## 4-byte Reload
-; CHECK-NEXT: addl %ecx, %esi
-; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edx
-; CHECK-NEXT: addl %esi, %edx
-; CHECK-NEXT: jmp LBB0_23
-; CHECK-NEXT: LBB0_19: ## %bb29
-; CHECK-NEXT: testl %edx, %edx
-; CHECK-NEXT: jle LBB0_22
-; CHECK-NEXT: ## %bb.20: ## %bb.nph11
-; CHECK-NEXT: leal 15(%esi), %eax
+; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx ## 4-byte Reload
+; CHECK-NEXT: addl %ecx, %edx
+; CHECK-NEXT: addl %edx, %esi
+; CHECK-NEXT: movl %ecx, %eax
+; CHECK-NEXT: shrl $31, %eax
+; CHECK-NEXT: addl %ecx, %eax
+; CHECK-NEXT: sarl %eax
+; CHECK-NEXT: subl $4, %esp
+; CHECK-NEXT: pushl %eax
+; CHECK-NEXT: pushl $128
+; CHECK-NEXT: pushl %esi
+; CHECK-NEXT: jmp LBB0_19
+; CHECK-NEXT: LBB0_20: ## %bb29
+; CHECK-NEXT: testl %ebx, %ebx
+; CHECK-NEXT: jle LBB0_23
+; CHECK-NEXT: ## %bb.21: ## %bb.nph11
+; CHECK-NEXT: leal 15(%ebp), %eax
; CHECK-NEXT: andl $-16, %eax
; CHECK-NEXT: movl %eax, {{[-0-9]+}}(%e{{[sb]}}p) ## 4-byte Spill
-; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edi
+; CHECK-NEXT: movl {{[0-9]+}}(%esp), %esi
; CHECK-NEXT: .p2align 4
-; CHECK-NEXT: LBB0_21: ## %bb30
+; CHECK-NEXT: LBB0_22: ## %bb30
; CHECK-NEXT: ## =>This Inner Loop Header: Depth=1
; CHECK-NEXT: subl $4, %esp
-; CHECK-NEXT: pushl %esi
-; CHECK-NEXT: pushl %ebx
+; CHECK-NEXT: pushl %ebp
; CHECK-NEXT: pushl %edi
-; CHECK-NEXT: movl %ebx, %ebp
-; CHECK-NEXT: movl %edx, %ebx
+; CHECK-NEXT: pushl %esi
; CHECK-NEXT: calll _memcpy
-; CHECK-NEXT: movl %ebx, %edx
-; CHECK-NEXT: movl %ebp, %ebx
; CHECK-NEXT: addl $16, %esp
-; CHECK-NEXT: addl %esi, %edi
-; CHECK-NEXT: addl {{[-0-9]+}}(%e{{[sb]}}p), %ebx ## 4-byte Folded Reload
-; CHECK-NEXT: decl %edx
-; CHECK-NEXT: jne LBB0_21
-; CHECK-NEXT: LBB0_22: ## %bb33
-; CHECK-NEXT: movl (%esp), %ecx ## 4-byte Reload
+; CHECK-NEXT: addl %ebp, %esi
+; CHECK-NEXT: addl {{[-0-9]+}}(%e{{[sb]}}p), %edi ## 4-byte Folded Reload
+; CHECK-NEXT: decl %ebx
+; CHECK-NEXT: jne LBB0_22
+; CHECK-NEXT: LBB0_23: ## %bb33
+; CHECK-NEXT: movl {{[-0-9]+}}(%e{{[sb]}}p), %ecx ## 4-byte Reload
; CHECK-NEXT: movl {{[0-9]+}}(%esp), %edx
; CHECK-NEXT: addl %ecx, %edx
-; CHECK-NEXT: LBB0_23: ## %bb33
; CHECK-NEXT: movl %ecx, %eax
; CHECK-NEXT: shrl $31, %eax
; CHECK-NEXT: addl %ecx, %eax
@@ -214,8 +227,9 @@ define void @foo(ptr %r, i32 %s, i32 %w, i32 %x, ptr %j, i32 %d) nounwind {
; CHECK-NEXT: pushl %eax
; CHECK-NEXT: pushl $128
; CHECK-NEXT: pushl %edx
+; CHECK-NEXT: LBB0_19: ## %bb26
; CHECK-NEXT: calll _memset
-; CHECK-NEXT: addl $44, %esp
+; CHECK-NEXT: addl $60, %esp
; CHECK-NEXT: LBB0_25: ## %return
; CHECK-NEXT: popl %esi
; CHECK-NEXT: popl %edi
@@ -223,7 +237,7 @@ define void @foo(ptr %r, i32 %s, i32 %w, i32 %x, ptr %j, i32 %d) nounwind {
; CHECK-NEXT: popl %ebp
; CHECK-NEXT: retl
; CHECK-NEXT: LBB0_24: ## %return
-; CHECK-NEXT: addl $28, %esp
+; CHECK-NEXT: addl $44, %esp
; CHECK-NEXT: jmp LBB0_25
entry:
%0 = mul i32 %x, %w
diff --git a/llvm/test/Transforms/Attributor/IPConstantProp/PR16052.ll b/llvm/test/Transforms/Attributor/IPConstantProp/PR16052.ll
index 6641fdb9b4ffe..b8e7df86ad664 100644
--- a/llvm/test/Transforms/Attributor/IPConstantProp/PR16052.ll
+++ b/llvm/test/Transforms/Attributor/IPConstantProp/PR16052.ll
@@ -19,7 +19,7 @@ define i64 @fn2() {
; CGSCC-NEXT: entry:
; CGSCC-NEXT: [[CONV:%.*]] = sext i32 undef to i64
; CGSCC-NEXT: [[DIV:%.*]] = sdiv i64 8, [[CONV]]
-; CGSCC-NEXT: [[CALL2:%.*]] = call i64 @fn1(i64 [[DIV]]) #[[ATTR2:[0-9]+]]
+; CGSCC-NEXT: [[CALL2:%.*]] = call range(i64 -8, 9) i64 @fn1(i64 [[DIV]]) #[[ATTR2:[0-9]+]]
; CGSCC-NEXT: ret i64 [[CALL2]]
;
entry:
@@ -45,7 +45,7 @@ define i64 @fn2b(i32 %arg) {
; CGSCC-NEXT: entry:
; CGSCC-NEXT: [[CONV:%.*]] = sext i32 [[ARG]] to i64
; CGSCC-NEXT: [[DIV:%.*]] = sdiv i64 8, [[CONV]]
-; CGSCC-NEXT: [[CALL2:%.*]] = call i64 @fn1(i64 [[DIV]]) #[[ATTR2]]
+; CGSCC-NEXT: [[CALL2:%.*]] = call range(i64 -8, 9) i64 @fn1(i64 [[DIV]]) #[[ATTR2]]
; CGSCC-NEXT: ret i64 [[CALL2]]
;
entry:
diff --git a/llvm/test/Transforms/LICM/update-scev-after-hoist.ll b/llvm/test/Transforms/LICM/update-scev-after-hoist.ll
index 1b99212be7c02..df78be6c36361 100644
--- a/llvm/test/Transforms/LICM/update-scev-after-hoist.ll
+++ b/llvm/test/Transforms/LICM/update-scev-after-hoist.ll
@@ -11,15 +11,15 @@ define i16 @main() {
; SCEV-EXPR-NEXT: %mul.n.reass.reass = mul i16 %mul, 8
; SCEV-EXPR-NEXT: --> (8 * %mul) U: [0,-7) S: [-32768,32761) Exits: -32768 LoopDispositions: { %loop: Variant }
; SCEV-EXPR-NEXT: %div.n = sdiv i16 %div, 2
-; SCEV-EXPR-NEXT: --> %div.n U: [-16384,16384) S: [-16384,16384) Exits: 3 LoopDispositions: { %loop: Variant }
+; SCEV-EXPR-NEXT: --> (%div /s 2) U: [-16384,16384) S: [-16384,16384) Exits: 3 LoopDispositions: { %loop: Variant }
; SCEV-EXPR-NEXT: %div.n.1 = sdiv i16 %div.n, 2
-; SCEV-EXPR-NEXT: --> %div.n.1 U: [-8192,8192) S: [-8192,8192) Exits: 1 LoopDispositions: { %loop: Variant }
+; SCEV-EXPR-NEXT: --> (%div /s 4) U: [-8192,8192) S: [-8192,8192) Exits: 1 LoopDispositions: { %loop: Variant }
; SCEV-EXPR-NEXT: %div.n.2 = sdiv i16 %div.n.1, 2
-; SCEV-EXPR-NEXT: --> %div.n.2 U: [-4096,4096) S: [-4096,4096) Exits: 0 LoopDispositions: { %loop: Variant }
+; SCEV-EXPR-NEXT: --> (%div /s 8) U: [-4096,4096) S: [-4096,4096) Exits: 0 LoopDispositions: { %loop: Variant }
; SCEV-EXPR-NEXT: %mul.n.3 = mul i16 %mul.n.reass.reass, 2
; SCEV-EXPR-NEXT: --> (16 * %mul) U: [0,-15) S: [-32768,32753) Exits: 0 LoopDispositions: { %loop: Variant }
; SCEV-EXPR-NEXT: %div.n.3 = sdiv i16 %div.n.2, 2
-; SCEV-EXPR-NEXT: --> %div.n.3 U: [-2048,2048) S: [-2048,2048) Exits: 0 LoopDispositions: { %loop: Variant }
+; SCEV-EXPR-NEXT: --> (%div /s 16) U: [-2048,2048) S: [-2048,2048) Exits: 0 LoopDispositions: { %loop: Variant }
; SCEV-EXPR-NEXT: %mul.lcssa = phi i16 [ %mul.n.reass.reass, %loop ]
; SCEV-EXPR-NEXT: --> (8 * %mul) U: [0,-7) S: [-32768,32761) --> -32768 U: [-32768,-32767) S: [-32768,-32767)
; SCEV-EXPR-NEXT: Determining loop execution counts for: @main
diff --git a/llvm/test/Transforms/LoopVectorize/trip-count-expansion-may-introduce-ub.ll b/llvm/test/Transforms/LoopVectorize/trip-count-expansion-may-introduce-ub.ll
index 662cf84b676bc..b25595c2f1d88 100644
--- a/llvm/test/Transforms/LoopVectorize/trip-count-expansion-may-introduce-ub.ll
+++ b/llvm/test/Transforms/LoopVectorize/trip-count-expansion-may-introduce-ub.ll
@@ -1091,20 +1091,46 @@ define i64 @multi_exit_4_exit_count_with_sdiv_by_value_in_latch(ptr %dst, i64 %N
; CHECK-LABEL: define i64 @multi_exit_4_exit_count_with_sdiv_by_value_in_latch(
; CHECK-SAME: ptr [[DST:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: entry:
+; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 0)
+; CHECK-NEXT: [[TMP0:%.*]] = freeze i64 42
+; CHECK-NEXT: [[TMP1:%.*]] = freeze i64 [[N]]
+; CHECK-NEXT: [[TMP2:%.*]] = sdiv i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: [[SMAX1:%.*]] = call i64 @llvm.smax.i64(i64 [[TMP2]], i64 0)
+; CHECK-NEXT: [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[SMAX]], i64 [[SMAX1]])
+; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[UMIN]], 1
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ule i64 [[TMP3]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: [[TMP4:%.*]] = and i64 [[TMP3]], 3
+; CHECK-NEXT: [[TMP5:%.*]] = icmp eq i64 [[TMP4]], 0
+; CHECK-NEXT: [[TMP6:%.*]] = select i1 [[TMP5]], i64 4, i64 [[TMP4]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[TMP3]], [[TMP6]]
; CHECK-NEXT: br label [[LOOP_HEADER:%.*]]
-; CHECK: loop.header:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[LOOP_LATCH:%.*]] ]
+; CHECK: vector.body:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], [[LOOP_HEADER]] ]
; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[DST]], i64 [[IV]]
-; CHECK-NEXT: store i32 1, ptr [[GEP]], align 4
-; CHECK-NEXT: [[C_0:%.*]] = icmp slt i64 [[IV]], [[N]]
+; CHECK-NEXT: store <4 x i32> splat (i32 1), ptr [[GEP]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP8]], label [[MIDDLE_BLOCK:%.*]], label [[LOOP_HEADER]], !llvm.loop [[LOOP28:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: br label [[SCALAR_PH]]
+; CHECK: scalar.ph:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], [[MIDDLE_BLOCK]] ], [ 0, [[ENTRY:%.*]] ]
+; CHECK-NEXT: br label [[LOOP_HEADER1:%.*]]
+; CHECK: loop.header:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], [[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], [[LOOP_LATCH:%.*]] ]
+; CHECK-NEXT: [[GEP1:%.*]] = getelementptr inbounds i32, ptr [[DST]], i64 [[IV1]]
+; CHECK-NEXT: store i32 1, ptr [[GEP1]], align 4
+; CHECK-NEXT: [[C_0:%.*]] = icmp slt i64 [[IV1]], [[N]]
; CHECK-NEXT: br i1 [[C_0]], label [[LOOP_LATCH]], label [[EXIT:%.*]]
; CHECK: loop.latch:
-; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV1]], 1
; CHECK-NEXT: [[D:%.*]] = sdiv i64 42, [[N]]
-; CHECK-NEXT: [[C_1:%.*]] = icmp slt i64 [[IV]], [[D]]
-; CHECK-NEXT: br i1 [[C_1]], label [[LOOP_HEADER]], label [[EXIT]]
+; CHECK-NEXT: [[C_1:%.*]] = icmp slt i64 [[IV1]], [[D]]
+; CHECK-NEXT: br i1 [[C_1]], label [[LOOP_HEADER1]], label [[EXIT]], !llvm.loop [[LOOP29:![0-9]+]]
; CHECK: exit:
-; CHECK-NEXT: [[P:%.*]] = phi i64 [ 1, [[LOOP_HEADER]] ], [ 0, [[LOOP_LATCH]] ]
+; CHECK-NEXT: [[P:%.*]] = phi i64 [ 1, [[LOOP_HEADER1]] ], [ 0, [[LOOP_LATCH]] ]
; CHECK-NEXT: ret i64 [[P]]
;
entry:
@@ -1152,7 +1178,7 @@ define i64 @multi_exit_4_exit_count_with_udiv_by_value_in_latch1(ptr %dst, i64 %
; CHECK-NEXT: store <4 x i32> splat (i32 1), ptr [[TMP5]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[TMP7]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP28:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP7]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
; CHECK: middle.block:
; CHECK-NEXT: br label [[SCALAR_PH]]
; CHECK: scalar.ph:
@@ -1169,7 +1195,7 @@ define i64 @multi_exit_4_exit_count_with_udiv_by_value_in_latch1(ptr %dst, i64 %
; CHECK-NEXT: [[D:%.*]] = udiv i64 42, [[N]]
; CHECK-NEXT: [[X:%.*]] = sub i64 100, [[D]]
; CHECK-NEXT: [[C_1:%.*]] = icmp slt i64 [[IV]], [[D]]
-; CHECK-NEXT: br i1 [[C_1]], label [[LOOP_HEADER]], label [[EXIT]], !llvm.loop [[LOOP29:![0-9]+]]
+; CHECK-NEXT: br i1 [[C_1]], label [[LOOP_HEADER]], label [[EXIT]], !llvm.loop [[LOOP31:![0-9]+]]
; CHECK: exit:
; CHECK-NEXT: [[P:%.*]] = phi i64 [ 1, [[LOOP_HEADER]] ], [ 0, [[LOOP_LATCH]] ]
; CHECK-NEXT: ret i64 [[P]]
@@ -1263,7 +1289,7 @@ define i64 @multi_exit_count_with_udiv_by_value_in_latch_different_bounds_diviso
; CHECK-NEXT: store <4 x i32> splat (i32 1), ptr [[TMP6]], align 4
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; CHECK-NEXT: br i1 [[TMP8]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP30:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP8]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP32:![0-9]+]]
; CHECK: middle.block:
; CHECK-NEXT: br label [[SCALAR_PH]]
; CHECK: scalar.ph:
@@ -1279,7 +1305,7 @@ define i64 @multi_exit_count_with_udiv_by_value_in_latch_different_bounds_diviso
; CHECK-NEXT: [[IV_NEXT]] = add i64 [[IV]], 1
; CHECK-NEXT: [[D:%.*]] = udiv i64 42, [[M_1]]
; CHECK-NEXT: [[C_1:%.*]] = icmp slt i64 [[IV]], [[D]]
-; CHECK-NEXT: br i1 [[C_1]], label [[LOOP_HEADER]], label [[EXIT]], !llvm.loop [[LOOP31:![0-9]+]]
+; CHECK-NEXT: br i1 [[C_1]], label [[LOOP_HEADER]], label [[EXIT]], !llvm.loop [[LOOP33:![0-9]+]]
; CHECK: exit:
; CHECK-NEXT: [[P:%.*]] = phi i64 [ 1, [[LOOP_HEADER]] ], [ 0, [[LOOP_LATCH]] ]
; CHECK-NEXT: ret i64 [[P]]
@@ -1341,4 +1367,6 @@ exit:
; CHECK: [[LOOP29]] = distinct !{[[LOOP29]], [[META2]], [[META1]]}
; CHECK: [[LOOP30]] = distinct !{[[LOOP30]], [[META1]], [[META2]]}
; CHECK: [[LOOP31]] = distinct !{[[LOOP31]], [[META2]], [[META1]]}
+; CHECK: [[LOOP32]] = distinct !{[[LOOP32]], [[META1]], [[META2]]}
+; CHECK: [[LOOP33]] = distinct !{[[LOOP33]], [[META2]], [[META1]]}
;.
diff --git a/polly/include/polly/Support/SCEVAffinator.h b/polly/include/polly/Support/SCEVAffinator.h
index 5e149b50b189c..1223574b9b19e 100644
--- a/polly/include/polly/Support/SCEVAffinator.h
+++ b/polly/include/polly/Support/SCEVAffinator.h
@@ -122,6 +122,7 @@ class SCEVAffinator final : public llvm::SCEVVisitor<SCEVAffinator, PWACtx> {
PWACtx visitAddExpr(const llvm::SCEVAddExpr *E);
PWACtx visitMulExpr(const llvm::SCEVMulExpr *E);
PWACtx visitUDivExpr(const llvm::SCEVUDivExpr *E);
+ PWACtx visitSDivExpr(const llvm::SCEVSDivExpr *E);
PWACtx visitAddRecExpr(const llvm::SCEVAddRecExpr *E);
PWACtx visitSMaxExpr(const llvm::SCEVSMaxExpr *E);
PWACtx visitSMinExpr(const llvm::SCEVSMinExpr *E);
diff --git a/polly/lib/Support/SCEVAffinator.cpp b/polly/lib/Support/SCEVAffinator.cpp
index 07155486bd1e1..bf0b75ae9eb41 100644
--- a/polly/lib/Support/SCEVAffinator.cpp
+++ b/polly/lib/Support/SCEVAffinator.cpp
@@ -515,19 +515,14 @@ PWACtx SCEVAffinator::visitUDivExpr(const SCEVUDivExpr *Expr) {
return DividendPWAC;
}
-PWACtx SCEVAffinator::visitSDivInstruction(Instruction *SDiv) {
- assert(SDiv->getOpcode() == Instruction::SDiv && "Assumed SDiv instruction!");
-
- auto *Scope = getScope();
- auto *Divisor = SDiv->getOperand(1);
- const SCEV *DivisorSCEV = SE.getSCEVAtScope(Divisor, Scope);
- auto DivisorPWAC = visit(DivisorSCEV);
- assert(isa<SCEVConstant>(DivisorSCEV) &&
+PWACtx SCEVAffinator::visitSDivExpr(const SCEVSDivExpr *Expr) {
+ const SCEV *Dividend = Expr->getLHS();
+ const SCEV *Divisor = Expr->getRHS();
+ assert(isa<SCEVConstant>(Divisor) &&
"SDiv is no parameter but has a non-constant RHS.");
- auto *Dividend = SDiv->getOperand(0);
- const SCEV *DividendSCEV = SE.getSCEVAtScope(Dividend, Scope);
- auto DividendPWAC = visit(DividendSCEV);
+ auto DivisorPWAC = visit(Divisor);
+ auto DividendPWAC = visit(Dividend);
DividendPWAC = combine(DividendPWAC, DivisorPWAC, isl_pw_aff_tdiv_q);
return DividendPWAC;
}
@@ -554,8 +549,6 @@ PWACtx SCEVAffinator::visitUnknown(const SCEVUnknown *Expr) {
switch (I->getOpcode()) {
case Instruction::IntToPtr:
return visit(SE.getSCEVAtScope(I->getOperand(0), getScope()));
- case Instruction::SDiv:
- return visitSDivInstruction(I);
case Instruction::SRem:
return visitSRemInstruction(I);
default:
diff --git a/polly/lib/Support/SCEVValidator.cpp b/polly/lib/Support/SCEVValidator.cpp
index ad62406d85e7e..acef64febf56c 100644
--- a/polly/lib/Support/SCEVValidator.cpp
+++ b/polly/lib/Support/SCEVValidator.cpp
@@ -375,27 +375,24 @@ class SCEVValidator : public SCEVVisitor<SCEVValidator, ValidatorResult> {
}
ValidatorResult visitDivision(const SCEV *Dividend, const SCEV *Divisor,
- const SCEV *DivExpr,
- Instruction *SDiv = nullptr) {
-
+ const SCEVDivExpr *DivExpr) {
// First check if we might be able to model the division, thus if the
// divisor is constant. If so, check the dividend, otherwise check if
// the whole division can be seen as a parameter.
if (isa<SCEVConstant>(Divisor) && !Divisor->isZero())
return visit(Dividend);
- // For signed divisions use the SDiv instruction to check for a parameter
- // division, for unsigned divisions check the operands.
- if (SDiv)
- return visitGenericInst(SDiv, DivExpr);
+ if (isa<SCEVSDivExpr>(DivExpr) && DivExpr->mayTriggerUB(SE)) {
+ POLLY_DEBUG(dbgs() << "INVALID: signed division may trigger UB");
+ return ValidatorResult(SCEVType::INVALID);
+ }
ValidatorResult LHS = visit(Dividend);
ValidatorResult RHS = visit(Divisor);
if (LHS.isConstant() && RHS.isConstant())
return ValidatorResult(SCEVType::PARAM, DivExpr);
- POLLY_DEBUG(
- dbgs() << "INVALID: unsigned division of non-constant expressions");
+ POLLY_DEBUG(dbgs() << "INVALID: division of non-constant expressions");
return ValidatorResult(SCEVType::INVALID);
}
@@ -408,13 +405,10 @@ class SCEVValidator : public SCEVVisitor<SCEVValidator, ValidatorResult> {
return visitDivision(Dividend, Divisor, Expr);
}
- ValidatorResult visitSDivInstruction(Instruction *SDiv, const SCEV *Expr) {
- assert(SDiv->getOpcode() == Instruction::SDiv &&
- "Assumed SDiv instruction!");
-
- const SCEV *Dividend = SE.getSCEV(SDiv->getOperand(0));
- const SCEV *Divisor = SE.getSCEV(SDiv->getOperand(1));
- return visitDivision(Dividend, Divisor, Expr, SDiv);
+ ValidatorResult visitSDivExpr(const SCEVSDivExpr *Expr) {
+ const SCEV *Dividend = Expr->getLHS();
+ const SCEV *Divisor = Expr->getRHS();
+ return visitDivision(Dividend, Divisor, Expr);
}
ValidatorResult visitSRemInstruction(Instruction *SRem, const SCEV *S) {
@@ -451,8 +445,6 @@ class SCEVValidator : public SCEVVisitor<SCEVValidator, ValidatorResult> {
return visit(SE.getSCEVAtScope(I->getOperand(0), Scope));
case Instruction::Load:
return visitLoadInstruction(I, Expr);
- case Instruction::SDiv:
- return visitSDivInstruction(I, Expr);
case Instruction::SRem:
return visitSRemInstruction(I, Expr);
default:
@@ -562,8 +554,7 @@ class SCEVFindValues final {
Values.insert(Unknown->getValue());
Instruction *Inst = dyn_cast<Instruction>(Unknown->getValue());
- if (!Inst || (Inst->getOpcode() != Instruction::SRem &&
- Inst->getOpcode() != Instruction::SDiv))
+ if (!Inst || Inst->getOpcode() != Instruction::SRem)
return false;
const SCEV *Dividend = SE.getSCEV(Inst->getOperand(1));
diff --git a/polly/lib/Support/ScopHelper.cpp b/polly/lib/Support/ScopHelper.cpp
index 20473c231341f..ba5a9adcd97da 100644
--- a/polly/lib/Support/ScopHelper.cpp
+++ b/polly/lib/Support/ScopHelper.cpp
@@ -230,12 +230,12 @@ void polly::recordAssumption(polly::RecordedAssumptionsTy *RecordedAssumptions,
/// reference to the ScalarEvolution they belong to, so a mixup does not
/// immediately cause a crash but certainly is a violation of its interface.
///
-/// The SCEVExpander will __not__ generate any code for an existing SDiv/SRem
+/// The SCEVExpander will __not__ generate any code for an existing SRem
/// instruction but just use it, if it is referenced as a SCEVUnknown. We want
/// however to generate new code if the instruction is in the analyzed region
/// and we generate code outside/in front of that region. Hence, we generate the
-/// code for the SDiv/SRem operands in front of the analyzed region and then
-/// create a new SDiv/SRem operation there too.
+/// code for the SRem operands in front of the analyzed region and then
+/// create a new SRem operation there too.
struct ScopExpander final : SCEVVisitor<ScopExpander, const SCEV *> {
friend struct SCEVVisitor<ScopExpander, const SCEV *>;
@@ -341,8 +341,7 @@ struct ScopExpander final : SCEVVisitor<ScopExpander, const SCEV *> {
else
IP = RTCBB->getParent()->getEntryBlock().getTerminator()->getIterator();
- if (!Inst || (Inst->getOpcode() != Instruction::SRem &&
- Inst->getOpcode() != Instruction::SDiv))
+ if (!Inst || (Inst->getOpcode() != Instruction::SRem))
return visitGenericInst(E, Inst, IP);
const SCEV *LHSScev = GenSE.getSCEV(Inst->getOperand(0));
@@ -379,10 +378,20 @@ struct ScopExpander final : SCEVVisitor<ScopExpander, const SCEV *> {
}
const SCEV *visitUDivExpr(const SCEVUDivExpr *E) {
auto *RHSScev = visit(E->getRHS());
- if (!GenSE.isKnownNonZero(RHSScev))
+ if (E->mayTriggerUB(GenSE))
RHSScev = GenSE.getUMaxExpr(RHSScev, GenSE.getConstant(E->getType(), 1));
return GenSE.getUDivExpr(visit(E->getLHS()), RHSScev);
}
+ const SCEV *visitSDivExpr(const SCEVSDivExpr *E) {
+ auto *LHSScev = visit(E->getRHS());
+ auto *RHSScev = visit(E->getRHS());
+ // We would need to freeze LHS and RHS, but there are no corresponding SCEV
+ // expressions for freeze. Hence, we reject all sdiv expressions that may
+ // trigger UB.
+ assert(!E->mayTriggerUB(GenSE) &&
+ "SDiv that triggers UB should be invalid");
+ return GenSE.getSDivExpr(LHSScev, RHSScev);
+ }
const SCEV *visitAddExpr(const SCEVAddExpr *E) {
SmallVector<SCEVUse, 4> NewOps;
for (const SCEV *Op : E->operands())
diff --git a/polly/test/CodeGen/inner_scev_sdiv_2.ll b/polly/test/CodeGen/inner_scev_sdiv_2.ll
index 247c102834b25..573f347932706 100644
--- a/polly/test/CodeGen/inner_scev_sdiv_2.ll
+++ b/polly/test/CodeGen/inner_scev_sdiv_2.ll
@@ -1,11 +1,9 @@
; RUN: opt %loadNPMPolly -S '-passes=polly<no-default-opts>' < %s | FileCheck %s
;
; The SCEV expression in this test case refers to a sequence of sdiv
-; instructions, which are part of different bbs in the SCoP. When code
-; generating the parameter expressions, the code that is generated by the SCEV
-; expander has still references to the in-scop instructions, which was invalid.
+; instructions, which are part of different bbs in the SCoP.
;
-; CHECK: polly.start
+; CHECK: for.body.51
;
target triple = "x86_64-unknown-linux-gnu"
diff --git a/polly/test/CodeGen/scop_expander_insert_point.ll b/polly/test/CodeGen/scop_expander_insert_point.ll
index add605ae59471..6ec5298dcb50c 100644
--- a/polly/test/CodeGen/scop_expander_insert_point.ll
+++ b/polly/test/CodeGen/scop_expander_insert_point.ll
@@ -1,11 +1,6 @@
; RUN: opt %loadNPMPolly '-passes=polly<no-default-opts>' -S -polly-invariant-load-hoisting=true < %s | FileCheck %s
;
-; CHECK: entry:
-; CHECK-NEXT: %outvalue.141.phiops = alloca i64
-; CHECK-NEXT: %.preload.s2a = alloca i8
-; CHECK-NEXT: %divpolly = sdiv i32 undef, -1
-; CHECK-NEXT: %div = sdiv i32 undef, undef
-;
+; CHECK: for.body17
target datalayout = "e-m:e-i64:64-f80:128-n8:16:32:64-S128"
; Function Attrs: nounwind uwtable
diff --git a/polly/test/ScopInfo/nonaffine-buildMemoryAccess.ll b/polly/test/ScopInfo/nonaffine-buildMemoryAccess.ll
index a52aae0d59168..74f1f47a1445e 100644
--- a/polly/test/ScopInfo/nonaffine-buildMemoryAccess.ll
+++ b/polly/test/ScopInfo/nonaffine-buildMemoryAccess.ll
@@ -1,8 +1,10 @@
; RUN: opt %loadNPMPolly -polly-allow-nonaffine-loops '-passes=polly-custom<scops>' -polly-print-scops -disable-output < %s 2>&1 | FileCheck %s
;
-; CHECK: Domain :=
-; CHECK-NEXT: { Stmt_while_cond_i__TO__while_end_i[] };
-;
+; CHECK: 'Polly - Create polyhedral description of Scops' for region: 'while.cond.i => while.end.i' in function 'func':
+; CHECK-NEXT: Invalid Scop!
+; CHECK-NEXT: 'Polly - Create polyhedral description of Scops' for region: 'entry => <Function Return>' in function 'func':
+; CHECK-NEXT: Invalid Scop!
+
define i32 @func(i32 %param0, i32 %param1, ptr %param2) #3 {
entry:
>From de1a7f04112a678920aece2e7640ad8affb0ade6 Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Mon, 24 Aug 2026 09:42:21 +0100
Subject: [PATCH 2/4] [SCEV] Fix copy-paste error, thanks Antonio!
---
llvm/lib/Analysis/ScalarEvolution.cpp | 2 +-
.../test/Analysis/ScalarEvolution/implied-via-division.ll | 8 ++++----
2 files changed, 5 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Analysis/ScalarEvolution.cpp b/llvm/lib/Analysis/ScalarEvolution.cpp
index 53be5fee22fcf..f56edc53a6900 100644
--- a/llvm/lib/Analysis/ScalarEvolution.cpp
+++ b/llvm/lib/Analysis/ScalarEvolution.cpp
@@ -13071,7 +13071,7 @@ bool ScalarEvolution::isImpliedViaOperations(CmpPredicate Pred, const SCEV *LHS,
return true;
} else if (auto *LHSSDivExpr = dyn_cast<SCEVSDivExpr>(LHS)) {
SCEVUse LL = LHSSDivExpr->getOperand(0);
- SCEVUse LR = LHSSDivExpr->getOperand(0);
+ SCEVUse LR = LHSSDivExpr->getOperand(1);
// Rules for division.
// We are going to perform some comparisons with Denominator and its
diff --git a/llvm/test/Analysis/ScalarEvolution/implied-via-division.ll b/llvm/test/Analysis/ScalarEvolution/implied-via-division.ll
index 733ad98f4e3df..961f929fb21bb 100644
--- a/llvm/test/Analysis/ScalarEvolution/implied-via-division.ll
+++ b/llvm/test/Analysis/ScalarEvolution/implied-via-division.ll
@@ -31,9 +31,9 @@ define void @implied1_samesign(i32 %n) {
; Prove that (n > 1) ===> (n / 2 s> 0).
; CHECK-LABEL: 'implied1_samesign'
; CHECK-NEXT: Determining loop execution counts for: @implied1_samesign
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (1 smax (%n /s 2)))<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (-1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741822
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (1 smax (%n /s 2)))<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
@@ -514,9 +514,9 @@ define void @swapped_predicate(i32 %n) {
; Prove that (n s>= 1) ===> (0 s>= -n / 2).
; CHECK-LABEL: 'swapped_predicate'
; CHECK-NEXT: Determining loop execution counts for: @swapped_predicate
-; CHECK-NEXT: Loop %header: backedge-taken count is (-1 * (0 smin (-1 + (-1 * (%n /s 2))<nsw>)<nsw>))<nsw>
+; CHECK-NEXT: Loop %header: backedge-taken count is (1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: constant max backedge-taken count is i32 1073741824
-; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (-1 * (0 smin (-1 + (-1 * (%n /s 2))<nsw>)<nsw>))<nsw>
+; CHECK-NEXT: Loop %header: symbolic max backedge-taken count is (1 + (%n /s 2))<nsw>
; CHECK-NEXT: Loop %header: Trip multiple is 1
;
entry:
>From 7a3f478c04240cf05419b51c3f5bff9bf462e9cc Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Mon, 24 Aug 2026 14:05:54 +0100
Subject: [PATCH 3/4] [SCEV] Minor NFC restructing
---
llvm/lib/Analysis/ScalarEvolution.cpp | 10 ++++++----
1 file changed, 6 insertions(+), 4 deletions(-)
diff --git a/llvm/lib/Analysis/ScalarEvolution.cpp b/llvm/lib/Analysis/ScalarEvolution.cpp
index f56edc53a6900..996d288bdfa8d 100644
--- a/llvm/lib/Analysis/ScalarEvolution.cpp
+++ b/llvm/lib/Analysis/ScalarEvolution.cpp
@@ -5303,6 +5303,7 @@ static std::optional<BinaryOp> MatchBinaryOp(Value *V, const DataLayout &DL,
case Instruction::Sub:
case Instruction::Mul:
case Instruction::UDiv:
+ case Instruction::SDiv:
case Instruction::URem:
case Instruction::And:
case Instruction::AShr:
@@ -7714,6 +7715,7 @@ ScalarEvolution::getOperandsToCreate(Value *V, SmallVectorImpl<Value *> &Ops) {
}
case Instruction::Sub:
case Instruction::UDiv:
+ case Instruction::SDiv:
case Instruction::URem:
break;
case Instruction::AShr:
@@ -7755,7 +7757,6 @@ ScalarEvolution::getOperandsToCreate(Value *V, SmallVectorImpl<Value *> &Ops) {
}
return getUnknown(V);
- case Instruction::SDiv:
case Instruction::SRem:
Ops.push_back(U->getOperand(0));
Ops.push_back(U->getOperand(1));
@@ -7996,6 +7997,10 @@ const SCEV *ScalarEvolution::createSCEV(Value *V) {
LHS = getSCEV(BO->LHS);
RHS = getSCEV(BO->RHS);
return getUDivExpr(LHS, RHS);
+ case Instruction::SDiv:
+ LHS = getSCEV(BO->LHS);
+ RHS = getSCEV(BO->RHS);
+ return getSDivExpr(LHS, RHS);
case Instruction::URem:
LHS = getSCEV(BO->LHS);
RHS = getSCEV(BO->RHS);
@@ -8322,9 +8327,6 @@ const SCEV *ScalarEvolution::createSCEV(Value *V) {
// Just don't deal with inttoptr casts.
return getUnknown(V);
- case Instruction::SDiv:
- return getSDivExpr(getSCEV(U->getOperand(0)), getSCEV(U->getOperand(1)));
-
case Instruction::SRem:
// If both operands are non-negative, this is just an urem.
if (isKnownNonNegative(getSCEV(U->getOperand(0))) &&
>From 503bc8d7bd24f80f8df1c65b2b624fba9171009a Mon Sep 17 00:00:00 2001
From: Ramkumar Ramachandra <artagnon at tenstorrent.com>
Date: Tue, 25 Aug 2026 09:35:28 +0100
Subject: [PATCH 4/4] [SCEV] Fix expansion, thanks Antonio!
---
.../Utils/ScalarEvolutionExpander.cpp | 35 +++++++++++++++++--
.../trip-count-expansion-may-introduce-ub.ll | 6 ++--
2 files changed, 35 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp b/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
index b3c8676066ac3..37a39dde63760 100644
--- a/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
+++ b/llvm/lib/Transforms/Utils/ScalarEvolutionExpander.cpp
@@ -745,12 +745,41 @@ Value *SCEVExpander::visitUDivExpr(SCEVUseT<const SCEVUDivExpr *> S) {
}
Value *SCEVExpander::visitSDivExpr(SCEVUseT<const SCEVSDivExpr *> S) {
- Value *LHS = expand(S->getLHS());
- Value *RHS = expand(S->getRHS());
- if (S->mayTriggerUB(SE)) {
+ const SCEV *LHSExpr = S->getLHS();
+ const SCEV *RHSExpr = S->getRHS();
+ Value *LHS = expand(LHSExpr);
+ Value *RHS = expand(RHSExpr);
+ if (!ScalarEvolution::isGuaranteedNotToBePoison(LHSExpr))
LHS = Builder.CreateFreeze(LHS);
+ if (!ScalarEvolution::isGuaranteedNotToBePoison(RHSExpr))
RHS = Builder.CreateFreeze(RHS);
+
+ // Select RHS to avoid UB
+ if (!ScalarEvolution::isGuaranteedNotToBePoison(RHSExpr) ||
+ !SE.isKnownNonZero(RHSExpr)) {
+ Constant *Null = ConstantInt::getNullValue(LHS->getType());
+ Constant *One = ConstantInt::get(RHS->getType(), 1);
+ RHS = Builder.CreateSelect(Builder.CreateICmp(CmpInst::ICMP_EQ, RHS, Null),
+ One, RHS);
+ }
+
+ // Select LHS to avoid UB
+ if (!ScalarEvolution::isGuaranteedNotToBePoison(LHSExpr) ||
+ (SE.getSignedRangeMin(LHSExpr).isMinSignedValue() &&
+ !SE.isKnownPredicate(
+ CmpInst::ICMP_NE, RHSExpr,
+ SE.getConstant(S->getType(), -1, /*isSigned=*/true)))) {
+ Constant *SMin = ConstantInt::get(
+ LHS->getType(),
+ APInt::getSignedMinValue(LHS->getType()->getIntegerBitWidth()));
+ Constant *AllOnes = ConstantInt::getAllOnesValue(RHS->getType());
+ Constant *Null = ConstantInt::getNullValue(LHS->getType());
+ Value *SMinCond = Builder.CreateLogicalAnd(
+ Builder.CreateICmp(CmpInst::ICMP_EQ, LHS, SMin),
+ Builder.CreateICmp(CmpInst::ICMP_EQ, RHS, AllOnes));
+ LHS = Builder.CreateSelect(SMinCond, Null, LHS);
}
+
return InsertBinop(Instruction::SDiv, LHS, RHS, SCEV::FlagAnyWrap,
/*IsSafeToHoist=*/!S->mayTriggerUB(SE));
}
diff --git a/llvm/test/Transforms/LoopVectorize/trip-count-expansion-may-introduce-ub.ll b/llvm/test/Transforms/LoopVectorize/trip-count-expansion-may-introduce-ub.ll
index b25595c2f1d88..a3cd892b7e504 100644
--- a/llvm/test/Transforms/LoopVectorize/trip-count-expansion-may-introduce-ub.ll
+++ b/llvm/test/Transforms/LoopVectorize/trip-count-expansion-may-introduce-ub.ll
@@ -1092,9 +1092,9 @@ define i64 @multi_exit_4_exit_count_with_sdiv_by_value_in_latch(ptr %dst, i64 %N
; CHECK-SAME: ptr [[DST:%.*]], i64 [[N:%.*]]) {
; CHECK-NEXT: entry:
; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 0)
-; CHECK-NEXT: [[TMP0:%.*]] = freeze i64 42
-; CHECK-NEXT: [[TMP1:%.*]] = freeze i64 [[N]]
-; CHECK-NEXT: [[TMP2:%.*]] = sdiv i64 [[TMP0]], [[TMP1]]
+; CHECK-NEXT: [[TMP0:%.*]] = icmp eq i64 [[N]], 0
+; CHECK-NEXT: [[TMP1:%.*]] = select i1 [[TMP0]], i64 1, i64 [[N]]
+; CHECK-NEXT: [[TMP2:%.*]] = sdiv i64 42, [[TMP1]]
; CHECK-NEXT: [[SMAX1:%.*]] = call i64 @llvm.smax.i64(i64 [[TMP2]], i64 0)
; CHECK-NEXT: [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[SMAX]], i64 [[SMAX1]])
; CHECK-NEXT: [[TMP3:%.*]] = add nuw nsw i64 [[UMIN]], 1
More information about the llvm-commits
mailing list