[llvm] [VPlan] Move IV predicate handling to VPlan. (PR #192876)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Tue May 5 10:34:56 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/192876
>From e3a9033bdaa0529fe7e0da3785d89745f08eb557 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sat, 2 May 2026 22:21:24 +0100
Subject: [PATCH] [VPlan] Move IV predicate handling to VPlan.
---
llvm/include/llvm/Analysis/IVDescriptors.h | 17 +-
llvm/include/llvm/Analysis/ScalarEvolution.h | 12 +-
.../Vectorize/LoopVectorizationLegality.h | 7 +-
llvm/lib/Analysis/IVDescriptors.cpp | 38 ++--
llvm/lib/Analysis/ScalarEvolution.cpp | 23 ++-
.../Vectorize/LoopVectorizationLegality.cpp | 99 +--------
.../Transforms/Vectorize/LoopVectorize.cpp | 23 +++
llvm/lib/Transforms/Vectorize/VPlan.h | 6 +
.../Vectorize/VPlanConstruction.cpp | 67 ++++++
.../Transforms/Vectorize/VPlanTransforms.h | 10 +
.../VPlan/vplan-print-after-all.ll | 1 +
.../Transforms/LoopVectorize/induction.ll | 64 +++---
.../LoopVectorize/predicated-inductions.ll | 194 ++++++++++++------
13 files changed, 342 insertions(+), 219 deletions(-)
diff --git a/llvm/include/llvm/Analysis/IVDescriptors.h b/llvm/include/llvm/Analysis/IVDescriptors.h
index 05c58d1a20afb..0cb1517e59782 100644
--- a/llvm/include/llvm/Analysis/IVDescriptors.h
+++ b/llvm/include/llvm/Analysis/IVDescriptors.h
@@ -28,6 +28,7 @@ class Loop;
class PredicatedScalarEvolution;
class ScalarEvolution;
class SCEV;
+class SCEVPredicate;
class StoreInst;
/// These are the kinds of recurrences that we support.
@@ -400,10 +401,10 @@ class InductionDescriptor {
/// analysis, it can be passed through \p Expr. If the def-use chain
/// associated with the phi includes casts (that we know we can ignore
/// under proper runtime checks), they are passed through \p CastsToIgnore.
- LLVM_ABI static bool
- isInductionPHI(PHINode *Phi, const Loop *L, ScalarEvolution *SE,
- InductionDescriptor &D, const SCEV *Expr = nullptr,
- SmallVectorImpl<Instruction *> *CastsToIgnore = nullptr);
+ LLVM_ABI static bool isInductionPHI(
+ PHINode *Phi, const Loop *L, ScalarEvolution *SE, InductionDescriptor &D,
+ ArrayRef<const SCEVPredicate *> Preds = {}, const SCEV *Expr = nullptr,
+ SmallVectorImpl<Instruction *> *CastsToIgnore = nullptr);
/// Returns true if \p Phi is a floating point induction in the loop \p L.
/// If \p Phi is an induction, the induction descriptor \p D will contain
@@ -444,12 +445,16 @@ class InductionDescriptor {
/// SCEV overflow check.
ArrayRef<Instruction *> getCastInsts() const { return RedundantCasts; }
+ /// Returns the SCEV predicates associated with this induction.
+ ArrayRef<const SCEVPredicate *> getPredicates() const { return Predicates; }
+
private:
/// Private constructor - used by \c isInductionPHI and
/// \c getCanonicalIntInduction.
InductionDescriptor(Value *Start, InductionKind K, const SCEV *Step,
BinaryOperator *InductionBinOp = nullptr,
- SmallVectorImpl<Instruction *> *Casts = nullptr);
+ SmallVectorImpl<Instruction *> *Casts = nullptr,
+ ArrayRef<const SCEVPredicate *> Preds = {});
/// Start value.
TrackingVH<Value> StartValue;
@@ -462,6 +467,8 @@ class InductionDescriptor {
// Instructions used for type-casts of the induction variable,
// that are redundant when guarded with a runtime SCEV overflow check.
SmallVector<Instruction *, 2> RedundantCasts;
+ // SCEV predicates (overflow checks) needed for this induction.
+ SmallVector<const SCEVPredicate *, 2> Predicates;
};
} // end namespace llvm
diff --git a/llvm/include/llvm/Analysis/ScalarEvolution.h b/llvm/include/llvm/Analysis/ScalarEvolution.h
index 5c01da0855f66..c85d57aa270db 100644
--- a/llvm/include/llvm/Analysis/ScalarEvolution.h
+++ b/llvm/include/llvm/Analysis/ScalarEvolution.h
@@ -2631,8 +2631,11 @@ class PredicatedScalarEvolution {
/// Attempts to produce an AddRecExpr for V by adding additional SCEV
/// predicates. If we can't transform the expression into an AddRecExpr we
/// return nullptr and not add additional SCEV predicates to the current
- /// context.
- LLVM_ABI const SCEVAddRecExpr *getAsAddRec(Value *V);
+ /// context. If \p ExtraPreds is non-null, the required predicates are
+ /// collected there instead of being added to this context.
+ LLVM_ABI const SCEVAddRecExpr *
+ getAsAddRec(Value *V,
+ SmallVectorImpl<const SCEVPredicate *> *ExtraPreds = nullptr);
/// Proves that V doesn't overflow by adding SCEV predicate.
LLVM_ABI void setNoOverflow(Value *V,
@@ -2655,8 +2658,9 @@ class PredicatedScalarEvolution {
/// Check if \p AR1 and \p AR2 are equal, while taking into account
/// Equal predicates in Preds.
- LLVM_ABI bool areAddRecsEqualWithPreds(const SCEVAddRecExpr *AR1,
- const SCEVAddRecExpr *AR2) const;
+ LLVM_ABI bool areAddRecsEqualWithPreds(
+ const SCEVAddRecExpr *AR1, const SCEVAddRecExpr *AR2,
+ ArrayRef<const SCEVPredicate *> ExtraPreds = {}) const;
private:
/// Increments the version number of the predicate. This needs to be called
diff --git a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
index 49e8bb1e85526..0c79b59d0359f 100644
--- a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
+++ b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
@@ -656,8 +656,7 @@ class LoopVectorizationLegality {
/// Updates the vectorization state by adding \p Phi to the inductions list.
/// This can set \p Phi as the main induction of the loop if \p Phi is a
/// better choice for the main induction than the existing one.
- void addInductionPhi(PHINode *Phi, const InductionDescriptor &ID,
- SmallPtrSetImpl<Value *> &AllowedExit);
+ void addInductionPhi(PHINode *Phi, const InductionDescriptor &ID);
/// The loop that we evaluate.
Loop *TheLoop;
@@ -715,10 +714,6 @@ class LoopVectorizationLegality {
/// Holds the widest induction type encountered.
IntegerType *WidestIndTy = nullptr;
- /// Allowed outside users. This holds the variables that can be accessed from
- /// outside the loop.
- SmallPtrSet<Value *, 4> AllowedExit;
-
/// Vectorization requirements that will go through late-evaluation.
LoopVectorizationRequirements *Requirements;
diff --git a/llvm/lib/Analysis/IVDescriptors.cpp b/llvm/lib/Analysis/IVDescriptors.cpp
index 96d30fd1b1415..3ea86e79560e8 100644
--- a/llvm/lib/Analysis/IVDescriptors.cpp
+++ b/llvm/lib/Analysis/IVDescriptors.cpp
@@ -1355,7 +1355,8 @@ RecurrenceDescriptor::getReductionOpChain(PHINode *Phi, Loop *L) const {
InductionDescriptor::InductionDescriptor(Value *Start, InductionKind K,
const SCEV *Step, BinaryOperator *BOp,
- SmallVectorImpl<Instruction *> *Casts)
+ SmallVectorImpl<Instruction *> *Casts,
+ ArrayRef<const SCEVPredicate *> Preds)
: StartValue(Start), IK(K), Step(Step), InductionBinOp(BOp) {
assert(IK != IK_NoInduction && "Not an induction");
@@ -1384,6 +1385,7 @@ InductionDescriptor::InductionDescriptor(Value *Start, InductionKind K,
if (Casts)
llvm::append_range(RedundantCasts, *Casts);
+ llvm::append_range(Predicates, Preds);
}
InductionDescriptor
@@ -1486,11 +1488,11 @@ bool InductionDescriptor::isFPInductionPHI(PHINode *Phi, const Loop *TheLoop,
static bool getCastsForInductionPHI(PredicatedScalarEvolution &PSE,
const SCEVUnknown *PhiScev,
const SCEVAddRecExpr *AR,
- SmallVectorImpl<Instruction *> &CastInsts) {
+ SmallVectorImpl<Instruction *> &CastInsts,
+ ArrayRef<const SCEVPredicate *> Preds) {
assert(CastInsts.empty() && "CastInsts is expected to be empty.");
auto *PN = cast<PHINode>(PhiScev->getValue());
- assert(PSE.getSCEV(PN) == AR && "Unexpected phi node SCEV expression");
const Loop *L = AR->getLoop();
// Find any cast instructions that participate in the def-use chain of
@@ -1524,6 +1526,11 @@ static bool getCastsForInductionPHI(PredicatedScalarEvolution &PSE,
if (!Val)
return false;
+ // Build a predicate to rewrite SCEVs of values in the cast chain using the
+ // predicates needed for this induction.
+ ScalarEvolution &SE = *PSE.getSE();
+ SCEVUnionPredicate Pred(Preds, SE);
+
// Follow the def-use chain until the induction phi is reached.
// If on the way we encounter a Value that has the same SCEV Expr as the
// phi node, we can consider the instructions we visit from that point
@@ -1536,8 +1543,9 @@ static bool getCastsForInductionPHI(PredicatedScalarEvolution &PSE,
if (!Inst || !L->contains(Inst)) {
return false;
}
- auto *AddRec = dyn_cast<SCEVAddRecExpr>(PSE.getSCEV(Val));
- if (AddRec && PSE.areAddRecsEqualWithPreds(AddRec, AR))
+ auto *AddRec = dyn_cast<SCEVAddRecExpr>(
+ SE.rewriteUsingPredicate(SE.getSCEV(Val), L, Pred));
+ if (AddRec && PSE.areAddRecsEqualWithPreds(AddRec, AR, Preds))
InCastSequence = true;
if (InCastSequence) {
// Only the last instruction in the cast sequence is expected to have
@@ -1575,9 +1583,12 @@ bool InductionDescriptor::isInductionPHI(PHINode *Phi, const Loop *TheLoop,
const SCEV *PhiScev = PSE.getSCEV(Phi);
const auto *AR = dyn_cast<SCEVAddRecExpr>(PhiScev);
+ // Collect predicates needed to force the SCEV into an AddRecExpr.
+ SmallVector<const SCEVPredicate *, 2> Preds;
+
// We need this expression to be an AddRecExpr.
if (Assume && !AR)
- AR = PSE.getAsAddRec(Phi);
+ AR = PSE.getAsAddRec(Phi, &Preds);
if (!AR) {
LLVM_DEBUG(dbgs() << "LV: PHI is not a poly recurrence.\n");
@@ -1593,17 +1604,17 @@ bool InductionDescriptor::isInductionPHI(PHINode *Phi, const Loop *TheLoop,
// induction.
if (PhiScev != AR && SymbolicPhi) {
SmallVector<Instruction *, 2> Casts;
- if (getCastsForInductionPHI(PSE, SymbolicPhi, AR, Casts))
- return isInductionPHI(Phi, TheLoop, PSE.getSE(), D, AR, &Casts);
+ if (getCastsForInductionPHI(PSE, SymbolicPhi, AR, Casts, Preds))
+ return isInductionPHI(Phi, TheLoop, PSE.getSE(), D, Preds, AR, &Casts);
}
- return isInductionPHI(Phi, TheLoop, PSE.getSE(), D, AR);
+ return isInductionPHI(Phi, TheLoop, PSE.getSE(), D, Preds, AR);
}
bool InductionDescriptor::isInductionPHI(
PHINode *Phi, const Loop *TheLoop, ScalarEvolution *SE,
- InductionDescriptor &D, const SCEV *Expr,
- SmallVectorImpl<Instruction *> *CastsToIgnore) {
+ InductionDescriptor &D, ArrayRef<const SCEVPredicate *> Preds,
+ const SCEV *Expr, SmallVectorImpl<Instruction *> *CastsToIgnore) {
Type *PhiTy = Phi->getType();
// isSCEVable returns true for integer and pointer types.
if (!SE->isSCEVable(PhiTy))
@@ -1643,13 +1654,14 @@ bool InductionDescriptor::isInductionPHI(
BinaryOperator *BOp =
dyn_cast<BinaryOperator>(Phi->getIncomingValueForBlock(Latch));
D = InductionDescriptor(StartValue, IK_IntInduction, Step, BOp,
- CastsToIgnore);
+ CastsToIgnore, Preds);
return true;
}
assert(PhiTy->isPointerTy() && "The PHI must be a pointer");
// This allows induction variables w/non-constant steps.
- D = InductionDescriptor(StartValue, IK_PtrInduction, Step);
+ D = InductionDescriptor(StartValue, IK_PtrInduction, Step,
+ /*InductionBinOp=*/nullptr, /*Casts=*/nullptr, Preds);
return true;
}
diff --git a/llvm/lib/Analysis/ScalarEvolution.cpp b/llvm/lib/Analysis/ScalarEvolution.cpp
index 676292ebe0346..e85e8ea0cdc7e 100644
--- a/llvm/lib/Analysis/ScalarEvolution.cpp
+++ b/llvm/lib/Analysis/ScalarEvolution.cpp
@@ -5863,16 +5863,17 @@ ScalarEvolution::createAddRecFromPHIWithCasts(const SCEVUnknown *SymbolicPHI) {
// even when the following Equal predicate exists:
// "%step == (sext ix (trunc iy to ix) to iy)".
bool PredicatedScalarEvolution::areAddRecsEqualWithPreds(
- const SCEVAddRecExpr *AR1, const SCEVAddRecExpr *AR2) const {
+ const SCEVAddRecExpr *AR1, const SCEVAddRecExpr *AR2,
+ ArrayRef<const SCEVPredicate *> ExtraPreds) const {
if (AR1 == AR2)
return true;
- auto areExprsEqual = [&](const SCEV *Expr1, const SCEV *Expr2) -> bool {
- if (Expr1 != Expr2 &&
- !Preds->implies(SE.getEqualPredicate(Expr1, Expr2), SE) &&
- !Preds->implies(SE.getEqualPredicate(Expr2, Expr1), SE))
- return false;
- return true;
+ SCEVUnionPredicate ExtraPred(ExtraPreds, SE);
+ SCEVUnionPredicate AllPreds = Preds->getUnionWith(&ExtraPred, SE);
+ auto areExprsEqual = [&](const SCEV *Expr1, const SCEV *Expr2) {
+ return Expr1 == Expr2 ||
+ AllPreds.implies(SE.getEqualPredicate(Expr1, Expr2), SE) ||
+ AllPreds.implies(SE.getEqualPredicate(Expr2, Expr1), SE);
};
if (!areExprsEqual(AR1->getStart(), AR2->getStart()) ||
@@ -15597,7 +15598,8 @@ bool PredicatedScalarEvolution::hasNoOverflow(
return Flags == SCEVWrapPredicate::IncrementAnyWrap;
}
-const SCEVAddRecExpr *PredicatedScalarEvolution::getAsAddRec(Value *V) {
+const SCEVAddRecExpr *PredicatedScalarEvolution::getAsAddRec(
+ Value *V, SmallVectorImpl<const SCEVPredicate *> *ExtraPreds) {
const SCEV *Expr = this->getSCEV(V);
SmallVector<const SCEVPredicate *, 4> NewPreds;
auto *New = SE.convertSCEVToAddRecWithPredicates(Expr, &L, NewPreds);
@@ -15605,6 +15607,11 @@ const SCEVAddRecExpr *PredicatedScalarEvolution::getAsAddRec(Value *V) {
if (!New)
return nullptr;
+ if (ExtraPreds) {
+ ExtraPreds->append(NewPreds);
+ return New;
+ }
+
for (const auto *P : NewPreds)
addPredicate(*P);
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index bee08eeba9927..baca0b43217a2 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -51,17 +51,6 @@ static cl::opt<bool>
cl::desc("Allow enabling loop hints to reorder "
"FP operations during vectorization."));
-// TODO: Move size-based thresholds out of legality checking, make cost based
-// decisions instead of hard thresholds.
-static cl::opt<unsigned> VectorizeSCEVCheckThreshold(
- "vectorize-scev-check-threshold", cl::init(16), cl::Hidden,
- cl::desc("The maximum number of SCEV checks allowed."));
-
-static cl::opt<unsigned> PragmaVectorizeSCEVCheckThreshold(
- "pragma-vectorize-scev-check-threshold", cl::init(128), cl::Hidden,
- cl::desc("The maximum number of SCEV checks allowed with a "
- "vectorize(enable) pragma"));
-
static cl::opt<LoopVectorizeHints::ScalableForceKind>
ForceScalableVectorization(
"scalable-vectorization", cl::init(LoopVectorizeHints::SK_Unspecified),
@@ -430,25 +419,6 @@ static IntegerType *getWiderInductionTy(const DataLayout &DL, Type *Ty0,
return TyA->getScalarSizeInBits() > TyB->getScalarSizeInBits() ? TyA : TyB;
}
-/// Check that the instruction has outside loop users and is not an
-/// identified reduction variable.
-static bool hasOutsideLoopUser(const Loop *TheLoop, Instruction *Inst,
- SmallPtrSetImpl<Value *> &AllowedExit) {
- // Reductions, Inductions and non-header phis are allowed to have exit users. All
- // other instructions must not have external users.
- if (!AllowedExit.count(Inst))
- // Check that all of the users of the loop are inside the BB.
- for (User *U : Inst->users()) {
- Instruction *UI = cast<Instruction>(U);
- // This user may be a reduction exit value.
- if (!TheLoop->contains(UI)) {
- LLVM_DEBUG(dbgs() << "LV: Found an outside user for : " << *UI << '\n');
- return true;
- }
- }
- return false;
-}
-
/// Returns true if A and B have same pointer operands or same SCEVs addresses
static bool storeToSameAddress(ScalarEvolution *SE, StoreInst *A,
StoreInst *B) {
@@ -691,9 +661,8 @@ bool LoopVectorizationLegality::canVectorizeOuterLoop() {
return Result;
}
-void LoopVectorizationLegality::addInductionPhi(
- PHINode *Phi, const InductionDescriptor &ID,
- SmallPtrSetImpl<Value *> &AllowedExit) {
+void LoopVectorizationLegality::addInductionPhi(PHINode *Phi,
+ const InductionDescriptor &ID) {
Inductions[Phi] = ID;
// In case this induction also comes with casts that we know we can ignore
@@ -732,17 +701,6 @@ void LoopVectorizationLegality::addInductionPhi(
PrimaryInduction = Phi;
}
- // Both the PHI node itself, and the "post-increment" value feeding
- // back into the PHI node may have external users.
- // We can allow those uses, except if the SCEVs we have for them rely
- // on predicates that only hold within the loop, since allowing the exit
- // currently means re-using this SCEV outside the loop (see PR33706 for more
- // details).
- if (PSE.getPredicate().isAlwaysTrue()) {
- AllowedExit.insert(Phi);
- AllowedExit.insert(Phi->getIncomingValueForBlock(TheLoop->getLoopLatch()));
- }
-
LLVM_DEBUG(dbgs() << "LV: Found an induction variable.\n");
}
@@ -754,7 +712,7 @@ bool LoopVectorizationLegality::setupOuterLoopInductions() {
InductionDescriptor ID;
if (InductionDescriptor::isInductionPHI(&Phi, TheLoop, PSE, ID) &&
ID.getKind() == InductionDescriptor::IK_IntInduction) {
- addInductionPhi(&Phi, ID, AllowedExit);
+ addInductionPhi(&Phi, ID);
return true;
}
// Bail out for any Phi in the outer loop header that is not a supported
@@ -862,7 +820,6 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
// Unsafe cyclic dependencies with header phis are identified during
// legalization for reduction, induction and fixed order
// recurrences.
- AllowedExit.insert(&I);
return true;
}
@@ -879,7 +836,6 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
if (RecurrenceDescriptor::isReductionPHI(Phi, TheLoop, RedDes, DB, AC, DT,
PSE.getSE())) {
Requirements->addExactFPMathInst(RedDes.getExactFPMathInst());
- AllowedExit.insert(RedDes.getLoopExitInstr());
Reductions[Phi] = std::move(RedDes);
assert((!RedDes.hasUsesOutsideReductionChain() ||
RecurrenceDescriptor::isMinMaxRecurrenceKind(
@@ -901,30 +857,21 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
ID.getConstIntStepValue() == nullptr;
};
- // TODO: Instead of recording the AllowedExit, it would be good to
- // record the complementary set: NotAllowedExit. These include (but may
- // not be limited to):
// 1. Reduction phis as they represent the one-before-last value, which
// is not available when vectorized
// 2. Induction phis and increment when SCEV predicates cannot be used
- // outside the loop - see addInductionPhi
- // 3. Non-Phis with outside uses when SCEV predicates cannot be used
- // outside the loop - see call to hasOutsideLoopUser in the non-phi
- // handling below
- // 4. FixedOrderRecurrence phis that can possibly be handled by
+ // outside the loop - see finalizeSCEVPredicates
+ // 3. FixedOrderRecurrence phis that can possibly be handled by
// extraction.
- // By recording these, we can then reason about ways to vectorize each
- // of these NotAllowedExit.
InductionDescriptor ID;
if (InductionDescriptor::isInductionPHI(Phi, TheLoop, PSE, ID) &&
!IsDisallowedStridedPointerInduction(ID)) {
- addInductionPhi(Phi, ID, AllowedExit);
+ addInductionPhi(Phi, ID);
Requirements->addExactFPMathInst(ID.getExactFPMathInst());
return true;
}
if (RecurrenceDescriptor::isFixedOrderRecurrence(Phi, TheLoop, DT)) {
- AllowedExit.insert(Phi);
FixedOrderRecurrences.insert(Phi);
return true;
}
@@ -933,7 +880,7 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
// and re-try classifying it a an induction PHI.
if (InductionDescriptor::isInductionPHI(Phi, TheLoop, PSE, ID, true) &&
!IsDisallowedStridedPointerInduction(ID)) {
- addInductionPhi(Phi, ID, AllowedExit);
+ addInductionPhi(Phi, ID);
return true;
}
@@ -1076,22 +1023,6 @@ bool LoopVectorizationLegality::canVectorizeInstr(Instruction &I) {
Hints->setPotentiallyUnsafe();
}
- // Reduction instructions are allowed to have exit users.
- // All other instructions must not have external users.
- if (hasOutsideLoopUser(TheLoop, &I, AllowedExit)) {
- // We can safely vectorize loops where instructions within the loop are
- // used outside the loop only if the SCEV predicates within the loop is
- // same as outside the loop. Allowing the exit means reusing the SCEV
- // outside the loop.
- if (PSE.getPredicate().isAlwaysTrue()) {
- AllowedExit.insert(&I);
- return true;
- }
- reportVectorizationFailure("Value cannot be used outside the loop",
- "ValueUsedOutsideLoop", ORE, TheLoop, &I);
- return false;
- }
-
return true;
}
@@ -2028,22 +1959,6 @@ bool LoopVectorizationLegality::canVectorize(bool UseVPlanNativePath) {
<< "!\n");
}
- unsigned SCEVThreshold = VectorizeSCEVCheckThreshold;
- if (Hints->getForce() == LoopVectorizeHints::FK_Enabled)
- SCEVThreshold = PragmaVectorizeSCEVCheckThreshold;
-
- if (PSE.getPredicate().getComplexity() > SCEVThreshold) {
- LLVM_DEBUG(dbgs() << "LV: Vectorization not profitable "
- "due to SCEVThreshold");
- reportVectorizationFailure("Too many SCEV checks needed",
- "Too many SCEV assumptions need to be made and checked at runtime",
- "TooManySCEVRunTimeChecks", ORE, TheLoop);
- if (DoExtraAnalysis)
- Result = false;
- else
- return false;
- }
-
// Okay! We've done all the tests. If any have failed, return false. Otherwise
// we can vectorize, and at this point we don't have any other mem analysis
// which may limit our maximum vectorization factor, so just return true with
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 8a770071752f3..03a88055cb1aa 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -297,6 +297,15 @@ cl::opt<unsigned> NumberOfStoresToPredicate(
"vectorize-num-stores-pred", cl::init(1), cl::Hidden,
cl::desc("Max number of stores to be predicated behind an if."));
+static cl::opt<unsigned> VectorizeSCEVCheckThreshold(
+ "vectorize-scev-check-threshold", cl::init(16), cl::Hidden,
+ cl::desc("The maximum number of SCEV checks allowed."));
+
+static cl::opt<unsigned> PragmaVectorizeSCEVCheckThreshold(
+ "pragma-vectorize-scev-check-threshold", cl::init(128), cl::Hidden,
+ cl::desc("The maximum number of SCEV checks allowed with a "
+ "vectorize(enable) pragma"));
+
static cl::opt<bool> EnableIndVarRegisterHeur(
"enable-ind-var-reg-heur", cl::init(true), cl::Hidden,
cl::desc("Count the induction variable only once when interleaving"));
@@ -6840,6 +6849,20 @@ void LoopVectorizationPlanner::buildVPlansWithVPRecipes(ElementCount MinVF,
RUN_VPLAN_PASS(VPlanTransforms::simplifyRecipes, *VPlan0);
RUN_VPLAN_PASS(VPlanTransforms::removeDeadRecipes, *VPlan0);
+
+ // Add surviving induction predicates to PSE and check constraints.
+ bool ForceVectorization = Hints.getForce() == LoopVectorizeHints::FK_Enabled;
+ bool NoEpilogueAllowed =
+ !ForceVectorization &&
+ (CM.EpilogueLoweringStatus == CM_EpilogueNotAllowedOptSize ||
+ CM.EpilogueLoweringStatus == CM_EpilogueNotAllowedLowTripLoop);
+ unsigned SCEVCheckThreshold = ForceVectorization
+ ? PragmaVectorizeSCEVCheckThreshold
+ : VectorizeSCEVCheckThreshold;
+ if (!RUN_VPLAN_PASS(VPlanTransforms::finalizeSCEVPredicates, *VPlan0, PSE,
+ NoEpilogueAllowed, SCEVCheckThreshold, ORE, OrigLoop))
+ return;
+
// If we're vectorizing a loop with an uncountable exit, make sure that the
// recipes are safe to handle.
// TODO: Remove this once we can properly check the VPlan itself for both
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 4a5420185224b..2c247e7bd546b 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -57,6 +57,7 @@ struct VPTransformState;
class raw_ostream;
class RecurrenceDescriptor;
class SCEV;
+class SCEVPredicate;
class Type;
class VPBasicBlock;
class VPBuilder;
@@ -2411,6 +2412,11 @@ class VPWidenInductionRecipe : public VPHeaderPHIRecipe {
/// Returns the induction descriptor for the recipe.
const InductionDescriptor &getInductionDescriptor() const { return IndDesc; }
+ /// Returns the SCEV predicates associated with this induction.
+ ArrayRef<const SCEVPredicate *> getPredicates() const {
+ return IndDesc.getPredicates();
+ }
+
VPValue *getBackedgeValue() override {
// TODO: All operands of base recipe must exist and be at same index in
// derived recipe.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index e20d5d947ac54..f3834dc41c56b 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -32,6 +32,7 @@
#include "llvm/Support/Debug.h"
#include "llvm/Transforms/Utils/LoopUtils.h"
#include "llvm/Transforms/Utils/LoopVersioning.h"
+#include "llvm/Transforms/Vectorize/LoopVectorize.h"
#define DEBUG_TYPE "vplan"
@@ -993,6 +994,72 @@ bool VPlanTransforms::createHeaderPhiRecipes(
return true;
}
+bool VPlanTransforms::finalizeSCEVPredicates(VPlan &Plan,
+ PredicatedScalarEvolution &PSE,
+ bool NoEpilogueAllowed,
+ unsigned SCEVCheckThreshold,
+ OptimizationRemarkEmitter *ORE,
+ Loop *TheLoop) {
+ // Collect which header IVs have predicates and add them to PSE.
+ auto *HeaderVPBB = cast<VPBasicBlock>(
+ Plan.getEntry()->getSuccessors()[1]->getSingleSuccessor());
+ SmallPtrSet<VPWidenInductionRecipe *, 4> PredicatedIVs;
+ for (auto &R : HeaderVPBB->phis()) {
+ auto *WideIV = dyn_cast<VPWidenInductionRecipe>(&R);
+ if (!WideIV || WideIV->getPredicates().empty())
+ continue;
+ PredicatedIVs.insert(WideIV);
+ for (const auto *P : WideIV->getPredicates())
+ PSE.addPredicate(*P);
+ }
+
+ // Bail out if exit phis use predicated IVs via ExitingIVValue, as the
+ // predicated SCEV may not hold outside the loop (PR33706). Check each IV's
+ // predicates directly, regardless of whether PSE was already non-trivial
+ // from other sources (e.g., LAI predicates).
+ // TODO: Overly conservative; exit values from vector registers are safe
+ // when guarded by runtime checks, but later passes may use the SCEV.
+ if (!PredicatedIVs.empty())
+ for (auto *EB : Plan.getExitBlocks())
+ for (VPRecipeBase &R : EB->phis())
+ for (VPValue *Op : R.operands()) {
+ VPValue *Inner;
+ if (!match(Op, m_ExitingIVValue(m_VPValue(Inner))))
+ continue;
+ auto *WideIV = dyn_cast<VPWidenInductionRecipe>(Inner);
+ if (!WideIV || !PredicatedIVs.contains(WideIV))
+ continue;
+ LLVM_DEBUG(dbgs() << "LV: Not vectorizing: Predicated IV has "
+ "outside-loop use via ExitingIVValue\n");
+ return false;
+ }
+
+ unsigned TotalComplexity = PSE.getPredicate().getComplexity();
+ if (TotalComplexity && NoEpilogueAllowed) {
+ LLVM_DEBUG(
+ dbgs() << "LV: Not vectorizing: SCEV predicates needed for induction "
+ "but optimizing for size\n");
+ reportVectorizationFailure(
+ "Runtime SCEV check is required with -Os/-Oz",
+ "runtime SCEV checks needed but optimizing for size",
+ "CantVersionLoopWithOptForSize", ORE, TheLoop);
+ return false;
+ }
+
+ if (TotalComplexity > SCEVCheckThreshold) {
+ LLVM_DEBUG(dbgs() << "LV: Not vectorizing: Too many SCEV checks needed ("
+ << TotalComplexity << " > " << SCEVCheckThreshold
+ << ")\n");
+ reportVectorizationFailure(
+ "Too many SCEV checks needed",
+ "Too many SCEV assumptions need to be made and checked at runtime",
+ "TooManySCEVRunTimeChecks", ORE, TheLoop);
+ return false;
+ }
+
+ return true;
+}
+
void VPlanTransforms::createInLoopReductionRecipes(VPlan &Plan,
ElementCount MinVF) {
VPTypeAnalysis TypeInfo(Plan);
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 6e11de399c406..db8d6632dd9c9 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -146,6 +146,16 @@ struct VPlanTransforms {
const SmallPtrSetImpl<const PHINode *> &FixedOrderRecurrences,
const SmallPtrSetImpl<PHINode *> &InLoopReductions, bool AllowReordering);
+ /// Finalize SCEV predicates by adding induction predicates from \p Plan to
+ /// \p PSE and checking constraints. Returns false if predicated IVs have
+ /// outside-loop uses via ExitingIVValue, if SCEV predicate complexity exceeds
+ /// \p SCEVCheckThreshold, or if predicates are needed but \p
+ /// NoEpilogueAllowed is true.
+ static bool
+ finalizeSCEVPredicates(VPlan &Plan, PredicatedScalarEvolution &PSE,
+ bool NoEpilogueAllowed, unsigned SCEVCheckThreshold,
+ OptimizationRemarkEmitter *ORE, Loop *TheLoop);
+
/// Create VPReductionRecipes for in-loop reductions. This processes chains
/// of operations contributing to in-loop reductions and creates appropriate
/// VPReductionRecipe instances.
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll
index 4bc9a8d96e542..40c5aced13e28 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-after-all.ll
@@ -7,6 +7,7 @@
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::createHeaderPhiRecipes
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::simplifyRecipes
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::removeDeadRecipes
+; CHECK: VPlan for loop in 'foo' after VPlanTransforms::finalizeSCEVPredicates
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::handleEarlyExits
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::addMiddleCheck
; CHECK: VPlan for loop in 'foo' after VPlanTransforms::createLoopRegions
diff --git a/llvm/test/Transforms/LoopVectorize/induction.ll b/llvm/test/Transforms/LoopVectorize/induction.ll
index 3b44b99b1ddeb..8d28e712cd6d7 100644
--- a/llvm/test/Transforms/LoopVectorize/induction.ll
+++ b/llvm/test/Transforms/LoopVectorize/induction.ll
@@ -3272,12 +3272,12 @@ define void @wrappingindvars1(i8 %t, i32 %len, ptr %A) {
; CHECK: vector.scevcheck:
; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[LEN]] to i8
; CHECK-NEXT: [[TMP2:%.*]] = add i8 [[T]], [[TMP1]]
-; CHECK-NEXT: [[TMP3:%.*]] = icmp ult i8 [[TMP2]], [[T]]
+; CHECK-NEXT: [[TMP3:%.*]] = icmp slt i8 [[TMP2]], [[T]]
; CHECK-NEXT: [[TMP4:%.*]] = icmp ugt i32 [[LEN]], 255
; CHECK-NEXT: [[TMP5:%.*]] = or i1 [[TMP3]], [[TMP4]]
; CHECK-NEXT: [[TMP6:%.*]] = trunc i32 [[LEN]] to i8
; CHECK-NEXT: [[TMP7:%.*]] = add i8 [[T]], [[TMP6]]
-; CHECK-NEXT: [[TMP8:%.*]] = icmp slt i8 [[TMP7]], [[T]]
+; CHECK-NEXT: [[TMP8:%.*]] = icmp ult i8 [[TMP7]], [[T]]
; CHECK-NEXT: [[TMP9:%.*]] = icmp ugt i32 [[LEN]], 255
; CHECK-NEXT: [[TMP10:%.*]] = or i1 [[TMP8]], [[TMP9]]
; CHECK-NEXT: [[TMP11:%.*]] = or i1 [[TMP5]], [[TMP10]]
@@ -3338,11 +3338,11 @@ define void @wrappingindvars1(i8 %t, i32 %len, ptr %A) {
; IND-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_SCEVCHECK:%.*]]
; IND: vector.scevcheck:
; IND-NEXT: [[TMP1:%.*]] = trunc i32 [[LEN]] to i8
-; IND-NEXT: [[TMP2:%.*]] = xor i8 [[T]], -1
-; IND-NEXT: [[TMP3:%.*]] = icmp ult i8 [[TMP2]], [[TMP1]]
+; IND-NEXT: [[TMP2:%.*]] = add i8 [[T]], [[TMP1]]
+; IND-NEXT: [[TMP3:%.*]] = icmp slt i8 [[TMP2]], [[T]]
; IND-NEXT: [[TMP4:%.*]] = trunc i32 [[LEN]] to i8
-; IND-NEXT: [[TMP5:%.*]] = add i8 [[T]], [[TMP4]]
-; IND-NEXT: [[TMP6:%.*]] = icmp slt i8 [[TMP5]], [[T]]
+; IND-NEXT: [[TMP5:%.*]] = xor i8 [[T]], -1
+; IND-NEXT: [[TMP6:%.*]] = icmp ult i8 [[TMP5]], [[TMP4]]
; IND-NEXT: [[TMP7:%.*]] = icmp ugt i32 [[LEN]], 255
; IND-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
; IND-NEXT: [[TMP9:%.*]] = or i1 [[TMP3]], [[TMP8]]
@@ -3404,11 +3404,11 @@ define void @wrappingindvars1(i8 %t, i32 %len, ptr %A) {
; UNROLL-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_SCEVCHECK:%.*]]
; UNROLL: vector.scevcheck:
; UNROLL-NEXT: [[TMP1:%.*]] = trunc i32 [[LEN]] to i8
-; UNROLL-NEXT: [[TMP2:%.*]] = xor i8 [[T]], -1
-; UNROLL-NEXT: [[TMP3:%.*]] = icmp ult i8 [[TMP2]], [[TMP1]]
+; UNROLL-NEXT: [[TMP2:%.*]] = add i8 [[T]], [[TMP1]]
+; UNROLL-NEXT: [[TMP3:%.*]] = icmp slt i8 [[TMP2]], [[T]]
; UNROLL-NEXT: [[TMP4:%.*]] = trunc i32 [[LEN]] to i8
-; UNROLL-NEXT: [[TMP5:%.*]] = add i8 [[T]], [[TMP4]]
-; UNROLL-NEXT: [[TMP6:%.*]] = icmp slt i8 [[TMP5]], [[T]]
+; UNROLL-NEXT: [[TMP5:%.*]] = xor i8 [[T]], -1
+; UNROLL-NEXT: [[TMP6:%.*]] = icmp ult i8 [[TMP5]], [[TMP4]]
; UNROLL-NEXT: [[TMP7:%.*]] = icmp ugt i32 [[LEN]], 255
; UNROLL-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
; UNROLL-NEXT: [[TMP9:%.*]] = or i1 [[TMP3]], [[TMP8]]
@@ -3475,12 +3475,12 @@ define void @wrappingindvars1(i8 %t, i32 %len, ptr %A) {
; UNROLL-NO-IC: vector.scevcheck:
; UNROLL-NO-IC-NEXT: [[TMP1:%.*]] = trunc i32 [[LEN]] to i8
; UNROLL-NO-IC-NEXT: [[TMP2:%.*]] = add i8 [[T]], [[TMP1]]
-; UNROLL-NO-IC-NEXT: [[TMP3:%.*]] = icmp ult i8 [[TMP2]], [[T]]
+; UNROLL-NO-IC-NEXT: [[TMP3:%.*]] = icmp slt i8 [[TMP2]], [[T]]
; UNROLL-NO-IC-NEXT: [[TMP4:%.*]] = icmp ugt i32 [[LEN]], 255
; UNROLL-NO-IC-NEXT: [[TMP5:%.*]] = or i1 [[TMP3]], [[TMP4]]
; UNROLL-NO-IC-NEXT: [[TMP6:%.*]] = trunc i32 [[LEN]] to i8
; UNROLL-NO-IC-NEXT: [[TMP7:%.*]] = add i8 [[T]], [[TMP6]]
-; UNROLL-NO-IC-NEXT: [[TMP8:%.*]] = icmp slt i8 [[TMP7]], [[T]]
+; UNROLL-NO-IC-NEXT: [[TMP8:%.*]] = icmp ult i8 [[TMP7]], [[T]]
; UNROLL-NO-IC-NEXT: [[TMP9:%.*]] = icmp ugt i32 [[LEN]], 255
; UNROLL-NO-IC-NEXT: [[TMP10:%.*]] = or i1 [[TMP8]], [[TMP9]]
; UNROLL-NO-IC-NEXT: [[TMP11:%.*]] = or i1 [[TMP5]], [[TMP10]]
@@ -3544,11 +3544,11 @@ define void @wrappingindvars1(i8 %t, i32 %len, ptr %A) {
; INTERLEAVE-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_SCEVCHECK:%.*]]
; INTERLEAVE: vector.scevcheck:
; INTERLEAVE-NEXT: [[TMP1:%.*]] = trunc i32 [[LEN]] to i8
-; INTERLEAVE-NEXT: [[TMP2:%.*]] = xor i8 [[T]], -1
-; INTERLEAVE-NEXT: [[TMP3:%.*]] = icmp ult i8 [[TMP2]], [[TMP1]]
+; INTERLEAVE-NEXT: [[TMP2:%.*]] = add i8 [[T]], [[TMP1]]
+; INTERLEAVE-NEXT: [[TMP3:%.*]] = icmp slt i8 [[TMP2]], [[T]]
; INTERLEAVE-NEXT: [[TMP4:%.*]] = trunc i32 [[LEN]] to i8
-; INTERLEAVE-NEXT: [[TMP5:%.*]] = add i8 [[T]], [[TMP4]]
-; INTERLEAVE-NEXT: [[TMP6:%.*]] = icmp slt i8 [[TMP5]], [[T]]
+; INTERLEAVE-NEXT: [[TMP5:%.*]] = xor i8 [[T]], -1
+; INTERLEAVE-NEXT: [[TMP6:%.*]] = icmp ult i8 [[TMP5]], [[TMP4]]
; INTERLEAVE-NEXT: [[TMP7:%.*]] = icmp ugt i32 [[LEN]], 255
; INTERLEAVE-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
; INTERLEAVE-NEXT: [[TMP9:%.*]] = or i1 [[TMP3]], [[TMP8]]
@@ -3647,12 +3647,12 @@ define void @wrappingindvars2(i8 %t, i32 %len, ptr %A) {
; CHECK: vector.scevcheck:
; CHECK-NEXT: [[TMP1:%.*]] = trunc i32 [[LEN]] to i8
; CHECK-NEXT: [[TMP2:%.*]] = add i8 [[T]], [[TMP1]]
-; CHECK-NEXT: [[TMP3:%.*]] = icmp ult i8 [[TMP2]], [[T]]
+; CHECK-NEXT: [[TMP3:%.*]] = icmp slt i8 [[TMP2]], [[T]]
; CHECK-NEXT: [[TMP4:%.*]] = icmp ugt i32 [[LEN]], 255
; CHECK-NEXT: [[TMP5:%.*]] = or i1 [[TMP3]], [[TMP4]]
; CHECK-NEXT: [[TMP6:%.*]] = trunc i32 [[LEN]] to i8
; CHECK-NEXT: [[TMP7:%.*]] = add i8 [[T]], [[TMP6]]
-; CHECK-NEXT: [[TMP8:%.*]] = icmp slt i8 [[TMP7]], [[T]]
+; CHECK-NEXT: [[TMP8:%.*]] = icmp ult i8 [[TMP7]], [[T]]
; CHECK-NEXT: [[TMP9:%.*]] = icmp ugt i32 [[LEN]], 255
; CHECK-NEXT: [[TMP10:%.*]] = or i1 [[TMP8]], [[TMP9]]
; CHECK-NEXT: [[TMP11:%.*]] = or i1 [[TMP5]], [[TMP10]]
@@ -3716,11 +3716,11 @@ define void @wrappingindvars2(i8 %t, i32 %len, ptr %A) {
; IND-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_SCEVCHECK:%.*]]
; IND: vector.scevcheck:
; IND-NEXT: [[TMP1:%.*]] = trunc i32 [[LEN]] to i8
-; IND-NEXT: [[TMP2:%.*]] = xor i8 [[T]], -1
-; IND-NEXT: [[TMP3:%.*]] = icmp ult i8 [[TMP2]], [[TMP1]]
+; IND-NEXT: [[TMP2:%.*]] = add i8 [[T]], [[TMP1]]
+; IND-NEXT: [[TMP3:%.*]] = icmp slt i8 [[TMP2]], [[T]]
; IND-NEXT: [[TMP4:%.*]] = trunc i32 [[LEN]] to i8
-; IND-NEXT: [[TMP5:%.*]] = add i8 [[T]], [[TMP4]]
-; IND-NEXT: [[TMP6:%.*]] = icmp slt i8 [[TMP5]], [[T]]
+; IND-NEXT: [[TMP5:%.*]] = xor i8 [[T]], -1
+; IND-NEXT: [[TMP6:%.*]] = icmp ult i8 [[TMP5]], [[TMP4]]
; IND-NEXT: [[TMP7:%.*]] = icmp ugt i32 [[LEN]], 255
; IND-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
; IND-NEXT: [[TMP9:%.*]] = or i1 [[TMP3]], [[TMP8]]
@@ -3785,11 +3785,11 @@ define void @wrappingindvars2(i8 %t, i32 %len, ptr %A) {
; UNROLL-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_SCEVCHECK:%.*]]
; UNROLL: vector.scevcheck:
; UNROLL-NEXT: [[TMP1:%.*]] = trunc i32 [[LEN]] to i8
-; UNROLL-NEXT: [[TMP2:%.*]] = xor i8 [[T]], -1
-; UNROLL-NEXT: [[TMP3:%.*]] = icmp ult i8 [[TMP2]], [[TMP1]]
+; UNROLL-NEXT: [[TMP2:%.*]] = add i8 [[T]], [[TMP1]]
+; UNROLL-NEXT: [[TMP3:%.*]] = icmp slt i8 [[TMP2]], [[T]]
; UNROLL-NEXT: [[TMP4:%.*]] = trunc i32 [[LEN]] to i8
-; UNROLL-NEXT: [[TMP5:%.*]] = add i8 [[T]], [[TMP4]]
-; UNROLL-NEXT: [[TMP6:%.*]] = icmp slt i8 [[TMP5]], [[T]]
+; UNROLL-NEXT: [[TMP5:%.*]] = xor i8 [[T]], -1
+; UNROLL-NEXT: [[TMP6:%.*]] = icmp ult i8 [[TMP5]], [[TMP4]]
; UNROLL-NEXT: [[TMP7:%.*]] = icmp ugt i32 [[LEN]], 255
; UNROLL-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
; UNROLL-NEXT: [[TMP9:%.*]] = or i1 [[TMP3]], [[TMP8]]
@@ -3859,12 +3859,12 @@ define void @wrappingindvars2(i8 %t, i32 %len, ptr %A) {
; UNROLL-NO-IC: vector.scevcheck:
; UNROLL-NO-IC-NEXT: [[TMP1:%.*]] = trunc i32 [[LEN]] to i8
; UNROLL-NO-IC-NEXT: [[TMP2:%.*]] = add i8 [[T]], [[TMP1]]
-; UNROLL-NO-IC-NEXT: [[TMP3:%.*]] = icmp ult i8 [[TMP2]], [[T]]
+; UNROLL-NO-IC-NEXT: [[TMP3:%.*]] = icmp slt i8 [[TMP2]], [[T]]
; UNROLL-NO-IC-NEXT: [[TMP4:%.*]] = icmp ugt i32 [[LEN]], 255
; UNROLL-NO-IC-NEXT: [[TMP5:%.*]] = or i1 [[TMP3]], [[TMP4]]
; UNROLL-NO-IC-NEXT: [[TMP6:%.*]] = trunc i32 [[LEN]] to i8
; UNROLL-NO-IC-NEXT: [[TMP7:%.*]] = add i8 [[T]], [[TMP6]]
-; UNROLL-NO-IC-NEXT: [[TMP8:%.*]] = icmp slt i8 [[TMP7]], [[T]]
+; UNROLL-NO-IC-NEXT: [[TMP8:%.*]] = icmp ult i8 [[TMP7]], [[T]]
; UNROLL-NO-IC-NEXT: [[TMP9:%.*]] = icmp ugt i32 [[LEN]], 255
; UNROLL-NO-IC-NEXT: [[TMP10:%.*]] = or i1 [[TMP8]], [[TMP9]]
; UNROLL-NO-IC-NEXT: [[TMP11:%.*]] = or i1 [[TMP5]], [[TMP10]]
@@ -3931,11 +3931,11 @@ define void @wrappingindvars2(i8 %t, i32 %len, ptr %A) {
; INTERLEAVE-NEXT: br i1 [[MIN_ITERS_CHECK]], label [[SCALAR_PH:%.*]], label [[VECTOR_SCEVCHECK:%.*]]
; INTERLEAVE: vector.scevcheck:
; INTERLEAVE-NEXT: [[TMP1:%.*]] = trunc i32 [[LEN]] to i8
-; INTERLEAVE-NEXT: [[TMP2:%.*]] = xor i8 [[T]], -1
-; INTERLEAVE-NEXT: [[TMP3:%.*]] = icmp ult i8 [[TMP2]], [[TMP1]]
+; INTERLEAVE-NEXT: [[TMP2:%.*]] = add i8 [[T]], [[TMP1]]
+; INTERLEAVE-NEXT: [[TMP3:%.*]] = icmp slt i8 [[TMP2]], [[T]]
; INTERLEAVE-NEXT: [[TMP4:%.*]] = trunc i32 [[LEN]] to i8
-; INTERLEAVE-NEXT: [[TMP5:%.*]] = add i8 [[T]], [[TMP4]]
-; INTERLEAVE-NEXT: [[TMP6:%.*]] = icmp slt i8 [[TMP5]], [[T]]
+; INTERLEAVE-NEXT: [[TMP5:%.*]] = xor i8 [[T]], -1
+; INTERLEAVE-NEXT: [[TMP6:%.*]] = icmp ult i8 [[TMP5]], [[TMP4]]
; INTERLEAVE-NEXT: [[TMP7:%.*]] = icmp ugt i32 [[LEN]], 255
; INTERLEAVE-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
; INTERLEAVE-NEXT: [[TMP9:%.*]] = or i1 [[TMP3]], [[TMP8]]
diff --git a/llvm/test/Transforms/LoopVectorize/predicated-inductions.ll b/llvm/test/Transforms/LoopVectorize/predicated-inductions.ll
index 6213831d00cab..1a7b5d336e9f5 100644
--- a/llvm/test/Transforms/LoopVectorize/predicated-inductions.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicated-inductions.ll
@@ -8,61 +8,137 @@
; RUN: -vectorize-scev-check-threshold=1 %s | FileCheck --check-prefixes=COMMON,THRESHOLD1 %s
define i64 @predicated_iv_with_liveout(ptr %dst, i64 %n) {
-; COMMON-LABEL: define i64 @predicated_iv_with_liveout(
-; COMMON-SAME: ptr [[DST:%.*]], i64 [[N:%.*]]) {
-; COMMON-NEXT: [[ENTRY:.*]]:
-; COMMON-NEXT: [[SMAX1:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; COMMON-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX1]], 4
-; COMMON-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
-; COMMON: [[VECTOR_SCEVCHECK]]:
-; COMMON-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
-; COMMON-NEXT: [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
-; COMMON-NEXT: [[TMP1:%.*]] = icmp ugt i64 [[TMP0]], 65535
-; COMMON-NEXT: br i1 [[TMP1]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
-; COMMON: [[VECTOR_PH]]:
-; COMMON-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[SMAX1]], 4
-; COMMON-NEXT: [[N_VEC:%.*]] = sub i64 [[SMAX1]], [[N_MOD_VF]]
-; COMMON-NEXT: [[TMP2:%.*]] = trunc i64 [[N_VEC]] to i16
-; COMMON-NEXT: br label %[[VECTOR_BODY:.*]]
-; COMMON: [[VECTOR_BODY]]:
-; COMMON-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; COMMON-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; COMMON-NEXT: [[VEC_IND2:%.*]] = phi <4 x i16> [ <i16 0, i16 1, i16 2, i16 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT3:%.*]], %[[VECTOR_BODY]] ]
-; COMMON-NEXT: [[VECTOR_RECUR:%.*]] = phi <4 x i64> [ <i64 poison, i64 poison, i64 poison, i64 0>, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
-; COMMON-NEXT: [[TMP3:%.*]] = add <4 x i16> [[VEC_IND2]], splat (i16 1)
-; COMMON-NEXT: [[TMP4]] = zext <4 x i16> [[TMP3]] to <4 x i64>
-; COMMON-NEXT: [[TMP5:%.*]] = shufflevector <4 x i64> [[VECTOR_RECUR]], <4 x i64> [[TMP4]], <4 x i32> <i32 3, i32 4, i32 5, i32 6>
-; COMMON-NEXT: [[TMP6:%.*]] = extractelement <4 x i64> [[TMP5]], i64 0
-; COMMON-NEXT: [[TMP7:%.*]] = getelementptr inbounds i64, ptr [[DST]], i64 [[TMP6]]
-; COMMON-NEXT: store <4 x i64> [[VEC_IND]], ptr [[TMP7]], align 8
-; COMMON-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
-; COMMON-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
-; COMMON-NEXT: [[VEC_IND_NEXT3]] = add <4 x i16> [[VEC_IND2]], splat (i16 4)
-; COMMON-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; COMMON-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; COMMON: [[MIDDLE_BLOCK]]:
-; COMMON-NEXT: [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <4 x i64> [[TMP4]], i64 3
-; COMMON-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[SMAX1]], [[N_VEC]]
-; COMMON-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
-; COMMON: [[SCALAR_PH]]:
-; COMMON-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_SCEVCHECK]] ]
-; COMMON-NEXT: [[BC_RESUME_VAL4:%.*]] = phi i16 [ [[TMP2]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_SCEVCHECK]] ]
-; COMMON-NEXT: [[SCALAR_RECUR_INIT:%.*]] = phi i64 [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_SCEVCHECK]] ]
-; COMMON-NEXT: br label %[[LOOP:.*]]
-; COMMON: [[LOOP]]:
-; COMMON-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; COMMON-NEXT: [[DEAD_IV:%.*]] = phi i16 [ [[BC_RESUME_VAL4]], %[[SCALAR_PH]] ], [ [[DEAD_IV_NEXT:%.*]], %[[LOOP]] ]
-; COMMON-NEXT: [[PREV:%.*]] = phi i64 [ [[SCALAR_RECUR_INIT]], %[[SCALAR_PH]] ], [ [[EXT:%.*]], %[[LOOP]] ]
-; COMMON-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
-; COMMON-NEXT: [[DEAD_IV_NEXT]] = add i16 [[DEAD_IV]], 1
-; COMMON-NEXT: [[EXT]] = zext i16 [[DEAD_IV_NEXT]] to i64
-; COMMON-NEXT: [[GEP:%.*]] = getelementptr inbounds i64, ptr [[DST]], i64 [[PREV]]
-; COMMON-NEXT: store i64 [[IV]], ptr [[GEP]], align 8
-; COMMON-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
-; COMMON-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
-; COMMON: [[EXIT]]:
-; COMMON-NEXT: [[RESULT:%.*]] = phi i64 [ [[EXT]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ]
-; COMMON-NEXT: ret i64 [[RESULT]]
+;
+; CHECK-LABEL: define i64 @predicated_iv_with_liveout(
+; CHECK-SAME: ptr [[DST:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[SMAX1:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX1]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; CHECK: [[VECTOR_SCEVCHECK]]:
+; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT: [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ugt i64 [[TMP0]], 65535
+; CHECK-NEXT: br i1 [[TMP1]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[SMAX1]], 4
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 [[SMAX1]], [[N_MOD_VF]]
+; CHECK-NEXT: [[TMP2:%.*]] = trunc i64 [[N_VEC]] to i16
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND2:%.*]] = phi <4 x i16> [ <i16 0, i16 1, i16 2, i16 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT3:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VECTOR_RECUR:%.*]] = phi <4 x i64> [ <i64 poison, i64 poison, i64 poison, i64 0>, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = add <4 x i16> [[VEC_IND2]], splat (i16 1)
+; CHECK-NEXT: [[TMP4]] = zext <4 x i16> [[TMP3]] to <4 x i64>
+; CHECK-NEXT: [[TMP5:%.*]] = shufflevector <4 x i64> [[VECTOR_RECUR]], <4 x i64> [[TMP4]], <4 x i32> <i32 3, i32 4, i32 5, i32 6>
+; CHECK-NEXT: [[TMP6:%.*]] = extractelement <4 x i64> [[TMP5]], i64 0
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr inbounds i64, ptr [[DST]], i64 [[TMP6]]
+; CHECK-NEXT: store <4 x i64> [[VEC_IND]], ptr [[TMP7]], align 8
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; CHECK-NEXT: [[VEC_IND_NEXT3]] = add <4 x i16> [[VEC_IND2]], splat (i16 4)
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <4 x i64> [[TMP4]], i64 3
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[SMAX1]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_SCEVCHECK]] ]
+; CHECK-NEXT: [[BC_RESUME_VAL4:%.*]] = phi i16 [ [[TMP2]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_SCEVCHECK]] ]
+; CHECK-NEXT: [[SCALAR_RECUR_INIT:%.*]] = phi i64 [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_SCEVCHECK]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[DEAD_IV:%.*]] = phi i16 [ [[BC_RESUME_VAL4]], %[[SCALAR_PH]] ], [ [[DEAD_IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[PREV:%.*]] = phi i64 [ [[SCALAR_RECUR_INIT]], %[[SCALAR_PH]] ], [ [[EXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[DEAD_IV_NEXT]] = add i16 [[DEAD_IV]], 1
+; CHECK-NEXT: [[EXT]] = zext i16 [[DEAD_IV_NEXT]] to i64
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i64, ptr [[DST]], i64 [[PREV]]
+; CHECK-NEXT: store i64 [[IV]], ptr [[GEP]], align 8
+; CHECK-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[RESULT:%.*]] = phi i64 [ [[EXT]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i64 [[RESULT]]
+;
+; THRESHOLD0-LABEL: define i64 @predicated_iv_with_liveout(
+; THRESHOLD0-SAME: ptr [[DST:%.*]], i64 [[N:%.*]]) {
+; THRESHOLD0-NEXT: [[ENTRY:.*]]:
+; THRESHOLD0-NEXT: br label %[[LOOP:.*]]
+; THRESHOLD0: [[LOOP]]:
+; THRESHOLD0-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; THRESHOLD0-NEXT: [[DEAD_IV:%.*]] = phi i16 [ 0, %[[ENTRY]] ], [ [[DEAD_IV_NEXT:%.*]], %[[LOOP]] ]
+; THRESHOLD0-NEXT: [[PREV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[EXT:%.*]], %[[LOOP]] ]
+; THRESHOLD0-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; THRESHOLD0-NEXT: [[DEAD_IV_NEXT]] = add i16 [[DEAD_IV]], 1
+; THRESHOLD0-NEXT: [[EXT]] = zext i16 [[DEAD_IV_NEXT]] to i64
+; THRESHOLD0-NEXT: [[GEP:%.*]] = getelementptr inbounds i64, ptr [[DST]], i64 [[PREV]]
+; THRESHOLD0-NEXT: store i64 [[IV]], ptr [[GEP]], align 8
+; THRESHOLD0-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; THRESHOLD0-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]]
+; THRESHOLD0: [[EXIT]]:
+; THRESHOLD0-NEXT: [[RESULT:%.*]] = phi i64 [ [[EXT]], %[[LOOP]] ]
+; THRESHOLD0-NEXT: ret i64 [[RESULT]]
+;
+; THRESHOLD1-LABEL: define i64 @predicated_iv_with_liveout(
+; THRESHOLD1-SAME: ptr [[DST:%.*]], i64 [[N:%.*]]) {
+; THRESHOLD1-NEXT: [[ENTRY:.*]]:
+; THRESHOLD1-NEXT: [[SMAX1:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; THRESHOLD1-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX1]], 4
+; THRESHOLD1-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
+; THRESHOLD1: [[VECTOR_SCEVCHECK]]:
+; THRESHOLD1-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; THRESHOLD1-NEXT: [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; THRESHOLD1-NEXT: [[TMP1:%.*]] = icmp ugt i64 [[TMP0]], 65535
+; THRESHOLD1-NEXT: br i1 [[TMP1]], label %[[SCALAR_PH]], label %[[VECTOR_PH:.*]]
+; THRESHOLD1: [[VECTOR_PH]]:
+; THRESHOLD1-NEXT: [[N_MOD_VF:%.*]] = urem i64 [[SMAX1]], 4
+; THRESHOLD1-NEXT: [[N_VEC:%.*]] = sub i64 [[SMAX1]], [[N_MOD_VF]]
+; THRESHOLD1-NEXT: [[TMP2:%.*]] = trunc i64 [[N_VEC]] to i16
+; THRESHOLD1-NEXT: br label %[[VECTOR_BODY:.*]]
+; THRESHOLD1: [[VECTOR_BODY]]:
+; THRESHOLD1-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; THRESHOLD1-NEXT: [[VEC_IND:%.*]] = phi <4 x i64> [ <i64 0, i64 1, i64 2, i64 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; THRESHOLD1-NEXT: [[VEC_IND2:%.*]] = phi <4 x i16> [ <i16 0, i16 1, i16 2, i16 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT3:%.*]], %[[VECTOR_BODY]] ]
+; THRESHOLD1-NEXT: [[VECTOR_RECUR:%.*]] = phi <4 x i64> [ <i64 poison, i64 poison, i64 poison, i64 0>, %[[VECTOR_PH]] ], [ [[TMP4:%.*]], %[[VECTOR_BODY]] ]
+; THRESHOLD1-NEXT: [[TMP3:%.*]] = add <4 x i16> [[VEC_IND2]], splat (i16 1)
+; THRESHOLD1-NEXT: [[TMP4]] = zext <4 x i16> [[TMP3]] to <4 x i64>
+; THRESHOLD1-NEXT: [[TMP5:%.*]] = shufflevector <4 x i64> [[VECTOR_RECUR]], <4 x i64> [[TMP4]], <4 x i32> <i32 3, i32 4, i32 5, i32 6>
+; THRESHOLD1-NEXT: [[TMP6:%.*]] = extractelement <4 x i64> [[TMP5]], i64 0
+; THRESHOLD1-NEXT: [[TMP7:%.*]] = getelementptr inbounds i64, ptr [[DST]], i64 [[TMP6]]
+; THRESHOLD1-NEXT: store <4 x i64> [[VEC_IND]], ptr [[TMP7]], align 8
+; THRESHOLD1-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; THRESHOLD1-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i64> [[VEC_IND]], splat (i64 4)
+; THRESHOLD1-NEXT: [[VEC_IND_NEXT3]] = add <4 x i16> [[VEC_IND2]], splat (i16 4)
+; THRESHOLD1-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; THRESHOLD1-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; THRESHOLD1: [[MIDDLE_BLOCK]]:
+; THRESHOLD1-NEXT: [[VECTOR_RECUR_EXTRACT:%.*]] = extractelement <4 x i64> [[TMP4]], i64 3
+; THRESHOLD1-NEXT: [[CMP_N:%.*]] = icmp eq i64 [[SMAX1]], [[N_VEC]]
+; THRESHOLD1-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; THRESHOLD1: [[SCALAR_PH]]:
+; THRESHOLD1-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_SCEVCHECK]] ]
+; THRESHOLD1-NEXT: [[BC_RESUME_VAL4:%.*]] = phi i16 [ [[TMP2]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_SCEVCHECK]] ]
+; THRESHOLD1-NEXT: [[SCALAR_RECUR_INIT:%.*]] = phi i64 [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ], [ 0, %[[VECTOR_SCEVCHECK]] ]
+; THRESHOLD1-NEXT: br label %[[LOOP:.*]]
+; THRESHOLD1: [[LOOP]]:
+; THRESHOLD1-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; THRESHOLD1-NEXT: [[DEAD_IV:%.*]] = phi i16 [ [[BC_RESUME_VAL4]], %[[SCALAR_PH]] ], [ [[DEAD_IV_NEXT:%.*]], %[[LOOP]] ]
+; THRESHOLD1-NEXT: [[PREV:%.*]] = phi i64 [ [[SCALAR_RECUR_INIT]], %[[SCALAR_PH]] ], [ [[EXT:%.*]], %[[LOOP]] ]
+; THRESHOLD1-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; THRESHOLD1-NEXT: [[DEAD_IV_NEXT]] = add i16 [[DEAD_IV]], 1
+; THRESHOLD1-NEXT: [[EXT]] = zext i16 [[DEAD_IV_NEXT]] to i64
+; THRESHOLD1-NEXT: [[GEP:%.*]] = getelementptr inbounds i64, ptr [[DST]], i64 [[PREV]]
+; THRESHOLD1-NEXT: store i64 [[IV]], ptr [[GEP]], align 8
+; THRESHOLD1-NEXT: [[CMP:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; THRESHOLD1-NEXT: br i1 [[CMP]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
+; THRESHOLD1: [[EXIT]]:
+; THRESHOLD1-NEXT: [[RESULT:%.*]] = phi i64 [ [[EXT]], %[[LOOP]] ], [ [[VECTOR_RECUR_EXTRACT]], %[[MIDDLE_BLOCK]] ]
+; THRESHOLD1-NEXT: ret i64 [[RESULT]]
;
entry:
br label %loop
@@ -336,6 +412,7 @@ define void @total_complexity_exceeds_threshold(ptr %dst, ptr %src, i64 %stride,
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX3]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
; CHECK: [[VECTOR_SCEVCHECK]]:
+; CHECK-NEXT: [[IDENT_CHECK:%.*]] = icmp ne i64 [[STRIDE]], 1
; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
; CHECK-NEXT: [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
; CHECK-NEXT: [[TMP1:%.*]] = trunc i64 [[TMP0]] to i8
@@ -343,8 +420,7 @@ define void @total_complexity_exceeds_threshold(ptr %dst, ptr %src, i64 %stride,
; CHECK-NEXT: [[MUL_OVERFLOW:%.*]] = extractvalue { i8, i1 } [[MUL]], 1
; CHECK-NEXT: [[TMP2:%.*]] = icmp ugt i64 [[TMP0]], 255
; CHECK-NEXT: [[TMP3:%.*]] = or i1 [[MUL_OVERFLOW]], [[TMP2]]
-; CHECK-NEXT: [[IDENT_CHECK:%.*]] = icmp ne i64 [[STRIDE]], 1
-; CHECK-NEXT: [[TMP4:%.*]] = or i1 [[TMP3]], [[IDENT_CHECK]]
+; CHECK-NEXT: [[TMP4:%.*]] = or i1 [[IDENT_CHECK]], [[TMP3]]
; CHECK-NEXT: br i1 [[TMP4]], label %[[SCALAR_PH]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[TMP5:%.*]] = sub i64 [[DST1]], [[SRC2]]
@@ -460,6 +536,7 @@ define void @combined_lai_iv_complexity(ptr %dst, ptr %src, i64 %stride, i64 %n)
; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[SMAX3]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_SCEVCHECK:.*]]
; CHECK: [[VECTOR_SCEVCHECK]]:
+; CHECK-NEXT: [[IDENT_CHECK:%.*]] = icmp ne i64 [[STRIDE]], 1
; CHECK-NEXT: [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
; CHECK-NEXT: [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
; CHECK-NEXT: [[TMP1:%.*]] = trunc i64 [[TMP0]] to i8
@@ -467,8 +544,7 @@ define void @combined_lai_iv_complexity(ptr %dst, ptr %src, i64 %stride, i64 %n)
; CHECK-NEXT: [[MUL_OVERFLOW:%.*]] = extractvalue { i8, i1 } [[MUL]], 1
; CHECK-NEXT: [[TMP2:%.*]] = icmp ugt i64 [[TMP0]], 255
; CHECK-NEXT: [[TMP3:%.*]] = or i1 [[MUL_OVERFLOW]], [[TMP2]]
-; CHECK-NEXT: [[IDENT_CHECK:%.*]] = icmp ne i64 [[STRIDE]], 1
-; CHECK-NEXT: [[TMP4:%.*]] = or i1 [[TMP3]], [[IDENT_CHECK]]
+; CHECK-NEXT: [[TMP4:%.*]] = or i1 [[IDENT_CHECK]], [[TMP3]]
; CHECK-NEXT: br i1 [[TMP4]], label %[[SCALAR_PH]], label %[[VECTOR_MEMCHECK:.*]]
; CHECK: [[VECTOR_MEMCHECK]]:
; CHECK-NEXT: [[TMP5:%.*]] = sub i64 [[DST1]], [[SRC2]]
More information about the llvm-commits
mailing list