[llvm] [LV] Vectorize uncountable early exit store loops with combined conditions (PR #205109)
Graham Hunter via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 23 06:28:18 PDT 2026
https://github.com/huntergr-arm updated https://github.com/llvm/llvm-project/pull/205109
>From ec3362ad2885be5d16c4f0d2471fd9522b839627 Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Fri, 5 Jun 2026 10:19:21 +0000
Subject: [PATCH 1/3] [LV] Vectorize uncountable early exit store loops with
combined conditions
Support the case where both the countable and uncountable exit conditions
have been combined by earlier passes.
---
.../Vectorize/LoopVectorizationLegality.h | 4 +
.../Vectorize/LoopVectorizationLegality.cpp | 150 +++++--
.../Transforms/Vectorize/LoopVectorize.cpp | 24 +-
.../Transforms/Vectorize/VPlanPatternMatch.h | 16 +-
.../Transforms/Vectorize/VPlanTransforms.cpp | 123 +++++-
.../Transforms/Vectorize/VPlanTransforms.h | 6 +
.../VPlan/early_exit_with_stores_vplan.ll | 71 ++++
.../VPlan/vplan-print-before-after-all.ll | 1 +
.../X86/vectorization-remarks-missed.ll | 18 +
.../early_exit_combined_exits.ll | 371 +++++++++++++++++-
.../early_exit_combined_exits_epilogue.ll | 73 ++++
.../early_exit_store_legality.ll | 7 +-
.../LoopVectorize/early_exit_with_stores.ll | 79 +++-
.../uncountable-single-exit-loops.ll | 2 +
...ountable-and-uncountable-exits-combined.ll | 40 +-
15 files changed, 926 insertions(+), 59 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/early_exit_combined_exits_epilogue.ll
diff --git a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
index 7b8b27c6541e15..c9eeae4b98b061 100644
--- a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
+++ b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
@@ -434,6 +434,10 @@ class LoopVectorizationLegality {
return getUncountableExitTrait() == UncountableExitTrait::ReadWrite;
}
+ /// If Cond is a combined exit condition featuring uncountable and countable
+ /// comparisons, returns the countable comparison. Otherwise returns nullptr.
+ Instruction *findCountableComparisonInCombinedCondition(Value *Cond) const;
+
/// Return true if there is store-load forwarding dependencies.
bool isSafeForAnyStoreLoadForwardDistances() const {
return LAI->getDepChecker().isSafeForAnyStoreLoadForwardDistances();
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index 953b1a41e9ee22..d9a93387d50148 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -22,6 +22,7 @@
#include "llvm/Analysis/MustExecute.h"
#include "llvm/Analysis/OptimizationRemarkEmitter.h"
#include "llvm/Analysis/ScalarEvolutionExpressions.h"
+#include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
#include "llvm/Analysis/TargetLibraryInfo.h"
#include "llvm/Analysis/TargetTransformInfo.h"
#include "llvm/Analysis/ValueTracking.h"
@@ -1649,6 +1650,69 @@ bool LoopVectorizationLegality::canVectorizeLoopNestCFG(
return Result;
}
+/// Matches an exit condition formed by comparing a value loaded from memory
+/// with another term. Binds the pointer, load, and the other comparison term.
+static bool matchUncountableExitCondition(Value *Cond, Value *&Ptr,
+ Instruction *&Load, Value *&Other) {
+ return match(Cond, m_OneUse(m_c_ICmp(
+ m_OneUse(m_Instruction(Load, m_Load(m_Value(Ptr)))),
+ m_Value(Other))));
+}
+
+/// Matches an exit condition formed by comparing the current value of a
+/// affine add recurrence in the given loop with a stride of 1 against a
+/// loop-invariant term.
+static bool matchCountableExitCondition(Value *Cond, ScalarEvolution &SE,
+ Loop *TheLoop) {
+ using namespace llvm::SCEVPatternMatch;
+ Value *IVUpdate, *Limit;
+ return match(Cond, m_c_ICmp(m_Value(IVUpdate, m_Add(m_Value(), m_Value())),
+ m_Value(Limit))) &&
+ TheLoop->isLoopInvariant(Limit) &&
+ SCEVPatternMatch::match(SE.getSCEV(IVUpdate),
+ m_scev_AffineAddRec(m_SCEV(), m_scev_One(),
+ m_SpecificLoop(TheLoop)));
+}
+
+/// Matches a combined exit condition consisting of an uncountable condition and
+/// a countable condition, combined by an or. Binds the pointer, load, the
+/// second comparison term for the uncountable condition, and the comparison for
+/// the countable condition.
+static bool matchCombinedExitCondition(Value *Cond, Instruction *&CountableCond,
+ Value *&Ptr, Instruction *&Load,
+ Value *&Other, ScalarEvolution &SE,
+ Loop *TheLoop) {
+ Value *L, *R;
+ if (!match(Cond, m_OneUse(m_LogicalOr(m_Value(L), m_Value(R)))))
+ return false;
+
+ if (matchCountableExitCondition(L, SE, TheLoop) &&
+ matchUncountableExitCondition(R, Ptr, Load, Other)) {
+ CountableCond = cast<Instruction>(L);
+ return true;
+ }
+
+ if (matchCountableExitCondition(R, SE, TheLoop) &&
+ matchUncountableExitCondition(L, Ptr, Load, Other)) {
+ CountableCond = cast<Instruction>(R);
+ return true;
+ }
+
+ return false;
+}
+
+Instruction *
+LoopVectorizationLegality::findCountableComparisonInCombinedCondition(
+ Value *Cond) const {
+ Value *Ptr, *Other;
+ Instruction *Load, *CountableCmp;
+ if (matchCombinedExitCondition(Cond, CountableCmp, Ptr, Load, Other,
+ *PSE.getSE(), TheLoop))
+ return CountableCmp;
+
+ return nullptr;
+}
+
bool LoopVectorizationLegality::isVectorizableEarlyExitLoop() {
BasicBlock *LatchBB = TheLoop->getLoopLatch();
if (!LatchBB) {
@@ -1660,9 +1724,9 @@ bool LoopVectorizationLegality::isVectorizableEarlyExitLoop() {
if (Reductions.size() || FixedOrderRecurrences.size()) {
reportVectorizationFailure(
- "Found reductions or recurrences in early-exit loop",
- "Cannot vectorize early exit loop with reductions or recurrences",
- "RecurrencesInEarlyExitLoop", ORE, TheLoop);
+ "Found reductions or recurrences in uncountable exit loop",
+ "Cannot vectorize uncountable exit loop with reductions or recurrences",
+ "RecurrencesInUncountableExitLoop", ORE, TheLoop);
return false;
}
@@ -1700,16 +1764,27 @@ bool LoopVectorizationLegality::isVectorizableEarlyExitLoop() {
}
// The latch block must have a countable exit.
- if (isa<SCEVCouldNotCompute>(
- PSE.getSE()->getPredicatedExitCount(TheLoop, LatchBB, &Predicates))) {
+ if (isa<SCEVCouldNotCompute>(PSE.getSE()->getPredicatedExitCount(
+ TheLoop, LatchBB, &Predicates, ScalarEvolution::SymbolicMaximum))) {
reportVectorizationFailure(
"Cannot determine exact exit count for latch block",
"Cannot vectorize early exit loop",
"UnknownLatchExitCountEarlyExitLoop", ORE, TheLoop);
return false;
}
- assert(llvm::is_contained(CountableExitingBlocks, LatchBB) &&
- "Latch block not found in list of countable exits!");
+
+ if (!is_contained(CountableExitingBlocks, LatchBB)) {
+ // If not a separate counted exit in the latch, then check for a combined
+ // countable and uncountable exit.
+ auto *Br = dyn_cast<CondBrInst>(LatchBB->getTerminator());
+ if (!Br ||
+ !findCountableComparisonInCombinedCondition(Br->getCondition())) {
+ reportVectorizationFailure(
+ "Latch block does not have a countable exit condition",
+ "NoCountableConditionInLatchBlock", ORE, TheLoop);
+ return false;
+ }
+ }
// Check to see if there are instructions that could potentially generate
// exceptions or have side-effects.
@@ -1787,6 +1862,13 @@ bool LoopVectorizationLegality::isVectorizableEarlyExitLoop() {
}
}
+ // We're only handling combined exit conditions via masking at present, which
+ // is used for loops with side effects.
+ // TODO: Support readonly loops with combined exit conditions.
+ // TODO: Decouple style from the presence of side effects.
+ if (!llvm::is_contained(CountableExitingBlocks, LatchBB) && !HasSideEffects)
+ return false;
+
[[maybe_unused]] const SCEV *SymbolicMaxBTC =
PSE.getSymbolicMaxBackedgeTakenCount();
// Since we have an exact exit count for the latch and the early exit
@@ -1813,20 +1895,26 @@ bool LoopVectorizationLegality::canUncountableExitConditionLoadBeMoved(
auto *Br = cast<CondBrInst>(ExitingBlock->getTerminator());
using namespace llvm::PatternMatch;
- Instruction *L = nullptr;
- Value *Ptr = nullptr;
- Value *R = nullptr;
- // The exit-condition load can appear on either side of the icmp.
- if (!match(Br->getCondition(),
- m_OneUse(m_c_ICmp(m_OneUse(m_Instruction(L, m_Load(m_Value(Ptr)))),
- m_Value(R))))) {
+ Value *Ptr, *Other;
+ Instruction *L, *CountableCond;
+ // We want to match either an uncounted condition (loaded value compared
+ // against a loop invariant value) or the combination (via logical or) of
+ // an uncounted condition with a counted condition (integer comparison of
+ // an induction variable for which we can identify an add recurrence within
+ // this loop).
+ if (!matchUncountableExitCondition(Br->getCondition(), Ptr, L, Other) &&
+ !matchCombinedExitCondition(Br->getCondition(), CountableCond, Ptr, L,
+ Other, *PSE.getSE(), TheLoop)) {
reportVectorizationFailure(
"Early exit loop with store but no supported condition load",
"NoConditionLoadForEarlyExitLoop", ORE, TheLoop);
return false;
}
- if (!TheLoop->isLoopInvariant(R)) {
+ // Bail if the uncountable exit load is compared against a non-invariant
+ // value.
+ // TODO: Remove this restriction.
+ if (!TheLoop->isLoopInvariant(Other)) {
reportVectorizationFailure(
"Early exit loop with store but no supported condition load",
"NoConditionLoadForEarlyExitLoop", ORE, TheLoop);
@@ -1944,24 +2032,17 @@ bool LoopVectorizationLegality::canVectorize(bool UseVPlanNativePath) {
return false;
}
- if (isa<SCEVCouldNotCompute>(PSE.getBackedgeTakenCount())) {
- if (TheLoop->getExitingBlock()) {
+ if (isa<SCEVCouldNotCompute>(PSE.getBackedgeTakenCount()) &&
+ !isVectorizableEarlyExitLoop()) {
+ assert(UncountableExitType == UncountableExitTrait::None &&
+ "Must be false without vectorizable early-exit loop");
+ if (TheLoop->getExitingBlock())
reportVectorizationFailure("Cannot vectorize uncountable loop",
"UnsupportedUncountableLoop", ORE, TheLoop);
- if (DoExtraAnalysis)
- Result = false;
- else
- return false;
- } else {
- if (!isVectorizableEarlyExitLoop()) {
- assert(UncountableExitType == UncountableExitTrait::None &&
- "Must be false without vectorizable early-exit loop");
- if (DoExtraAnalysis)
- Result = false;
- else
- return false;
- }
- }
+ if (DoExtraAnalysis)
+ Result = false;
+ else
+ return false;
}
// Go over each instruction and look at memory deps.
@@ -2011,6 +2092,13 @@ bool LoopVectorizationLegality::canFoldTailByMasking() const {
return false;
}
+ // TODO: Support tail folding with uncountable exits.
+ if (hasUncountableEarlyExit()) {
+ LLVM_DEBUG(dbgs() << "LV: Cannot tail fold by masking. Loop contains an "
+ "uncountable early exit.\n");
+ return false;
+ }
+
LLVM_DEBUG(dbgs() << "LV: checking if tail can be folded by masking.\n");
// The list of pointers that we can safely read and write to remains empty.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index d929af8afbd1da..43aceff4a7b8e7 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -2729,8 +2729,24 @@ void LoopVectorizationCostModel::collectLoopUniforms(ElementCount VF) {
if (Legal->hasUncountableEarlyExit() && TheLoop->getLoopLatch() != E)
continue;
auto *Cmp = dyn_cast<Instruction>(E->getTerminator()->getOperand(0));
- if (Cmp && TheLoop->contains(Cmp) && Cmp->hasOneUse())
- AddToWorklistIfAllowed(Cmp);
+ if (!Cmp || !TheLoop->contains(Cmp) || !Cmp->hasOneUse())
+ continue;
+
+ // If we have an exit condition that is actually two conditions (one
+ // countable and the other uncountable) combined via an or, only add the
+ // countable comparison as a uniform value.
+ if (Legal->hasUncountableExitWithSideEffects() &&
+ TheLoop->getLoopLatch() == E) {
+ if (Instruction *Countable =
+ Legal->findCountableComparisonInCombinedCondition(Cmp)) {
+ if (Countable->hasOneUse())
+ AddToWorklistIfAllowed(Countable);
+ continue;
+ }
+ }
+
+ // Normal exit comparisons are uniform.
+ AddToWorklistIfAllowed(Cmp);
}
auto PrevVF = VF.divideCoefficientBy(2);
@@ -6487,6 +6503,10 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
RUN_VPLAN_PASS(VPlanTransforms::addMiddleCheck, *VPlan0);
+ if (!RUN_VPLAN_PASS(VPlanTransforms::splitCombinedExits, *VPlan0, PSE,
+ OrigLoop))
+ return nullptr;
+
// If we're vectorizing a loop with an uncountable exit, make sure that the
// recipes are safe to handle.
// TODO: Remove this once we can properly check the VPlan itself for both
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
index a4bfab11055e6e..66416ab2023fad 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanPatternMatch.h
@@ -786,6 +786,12 @@ inline Cmp_match<Op0_t, Op1_t, Instruction::ICmp> m_ICmp(const Op0_t &Op0,
return Cmp_match<Op0_t, Op1_t, Instruction::ICmp>(Op0, Op1);
}
+template <typename Op0_t, typename Op1_t>
+inline auto m_c_ICmp(const Op0_t &Op0, const Op1_t &Op1) {
+ return m_CombineOr(Cmp_match<Op0_t, Op1_t, Instruction::ICmp>(Op0, Op1),
+ Cmp_match<Op1_t, Op0_t, Instruction::ICmp>(Op1, Op0));
+}
+
template <typename Op0_t, typename Op1_t>
inline Cmp_match<Op0_t, Op1_t, Instruction::ICmp>
m_ICmp(CmpPredicate &Pred, const Op0_t &Op0, const Op1_t &Op1) {
@@ -806,6 +812,13 @@ m_Cmp(const Op0_t &Op0, const Op1_t &Op1) {
Op1);
}
+template <typename Op0_t, typename Op1_t>
+inline auto m_c_Cmp(const Op0_t &Op0, const Op1_t &Op1) {
+ return m_CombineOr(
+ Cmp_match<Op0_t, Op1_t, Instruction::ICmp, Instruction::FCmp>(Op0, Op1),
+ Cmp_match<Op1_t, Op0_t, Instruction::ICmp, Instruction::FCmp>(Op1, Op0));
+}
+
template <typename Op0_t, typename Op1_t>
inline Cmp_match<Op0_t, Op1_t, Instruction::ICmp, Instruction::FCmp>
m_Cmp(CmpPredicate &Pred, const Op0_t &Op0, const Op1_t &Op1) {
@@ -923,7 +936,8 @@ inline auto m_LogicalOr(const Op0_t &Op0, const Op1_t &Op1) {
template <typename Op0_t, typename Op1_t>
inline auto m_c_LogicalOr(const Op0_t &Op0, const Op1_t &Op1) {
- return m_c_Select(Op0, m_True(), Op1);
+ return m_CombineOr(m_c_Select(Op0, m_True(), Op1),
+ m_c_Select(Op1, m_True(), Op0));
}
/// Match the canonical induction variable (IV) of any loop region.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 424a829407e0c2..8d61d7e54373d3 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -3005,6 +3005,121 @@ void VPlanTransforms::createInterleaveGroups(
}
}
+/// Matches an exit condition formed by comparing a value loaded from memory
+/// with a loop-invariant term. Binds the comparison for the condition.
+static auto m_Uncountable(VPValue *&Cond) {
+ return m_VPValue(
+ Cond,
+ m_c_Cmp(m_VPInstruction<Instruction::Load>(m_VPValue()), m_LiveIn()));
+}
+
+struct CountableConditionMatch {
+ VPValue *&Cmp;
+ PredicatedScalarEvolution &PSE;
+ Loop *L;
+
+ CountableConditionMatch(VPValue *&Cmp, PredicatedScalarEvolution &PSE,
+ Loop *L)
+ : Cmp(Cmp), PSE(PSE), L(L) {}
+
+ template <typename ITy> bool match(ITy *V) const {
+ VPValue *Update;
+ if (!VPlanPatternMatch::match(
+ V, m_VPValue(Cmp, m_c_ICmp(m_VPValue(Update, m_Add(m_VPValue(),
+ m_VPValue())),
+ m_LiveIn()))))
+ return false;
+
+ const SCEV *S = vputils::getSCEVExprForVPValue(Update, PSE, L);
+ return SCEVPatternMatch::match(
+ S, m_scev_AffineAddRec(m_SCEV(), m_scev_One(), m_SpecificLoop(L)));
+ }
+};
+
+/// Matches an exit condition formed by comparing the current value of a
+/// affine add recurrence in the given loop with a stride of 1 against a
+/// loop-invariant term. Binds the comparison for the condition.
+static auto m_Countable(VPValue *&Cmp, PredicatedScalarEvolution &PSE,
+ Loop *L) {
+ return CountableConditionMatch(Cmp, PSE, L);
+}
+
+bool VPlanTransforms::splitCombinedExits(VPlan &Plan,
+ PredicatedScalarEvolution &PSE,
+ Loop *L) {
+ // Check for a single combined exit in the latch block.
+ // TODO: Generalize to other blocks besides the latch.
+ // If we don't find a combined condition in the latch, just return true
+ // to proceed with vectorization.
+ auto [_, LatchVPBB] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
+ VPValue *Uncountable = nullptr;
+ VPValue *Countable = nullptr;
+ // We're looking for a conditional branch...
+ auto *Term = dyn_cast<VPInstruction>(LatchVPBB->getTerminator());
+ if (!Term || Term->getOpcode() != VPInstruction::BranchOnCond)
+ return true;
+
+ // ...where the condition is a combination of both a countable and an
+ // uncountable comparison.
+ VPValue *Cond = Term->getOperand(0);
+ if (!match(Cond, m_OneUse(m_CombineOr(
+ m_c_LogicalOr(m_Uncountable(Uncountable),
+ m_Countable(Countable, PSE, L)),
+ m_c_BinaryOr(m_Uncountable(Uncountable),
+ m_Countable(Countable, PSE, L))))))
+ return true;
+
+ // If the conditions are combined with a logical or (select), then we'll
+ // need to freeze the individual terms when splitting.
+ bool NeedsFreeze = match(Cond, m_LogicalOr(m_VPValue(), m_VPValue()));
+
+ // If we do have a combined exit condition, bail out if there's more than
+ // one exit block.
+ // TODO: Support additional exits.
+ ArrayRef<VPIRBasicBlock *> ExitBlocks = Plan.getExitBlocks();
+ if (ExitBlocks.size() != 1)
+ return false;
+
+ // If there are any live-outs, bail out. The exit block is an existing IR
+ // block, and if we split the exiting block then the incoming blocks and
+ // values won't be correct.
+ // TODO: Support live-outs with combined exits.
+ if (!ExitBlocks.front()->phis().empty())
+ return false;
+
+ // Split the latch block just before the terminator.
+ VPBasicBlock *NewLatch = LatchVPBB->splitAt(Term->getIterator());
+
+ // Create new terminator for uncountable condition.
+ VPBuilder EEBuilder(LatchVPBB);
+ if (NeedsFreeze)
+ Uncountable = EEBuilder.createScalarFreeze(
+ Uncountable, Uncountable->getScalarType(), DebugLoc());
+ EEBuilder.createNaryOp(VPInstruction::BranchOnCond, {Uncountable});
+
+ // We need to connect the uncountable exit to the sole exit block. The
+ // latch is expected to connect to the middle block instead.
+ // In canonical form, the backedge is the last successor for the latch. So
+ // the first successor (true path) should be the exit for both conditions.
+ LatchVPBB->clearSuccessors();
+ NewLatch->clearPredecessors();
+ VPBlockUtils::connectBlocks(LatchVPBB, ExitBlocks.front());
+ VPBlockUtils::connectBlocks(LatchVPBB, NewLatch);
+
+ // Set condition for latch block to countable condition.
+ if (NeedsFreeze) {
+ VPBuilder NewLatchBuilder(Term);
+ Countable = NewLatchBuilder.createScalarFreeze(
+ Countable, Countable->getScalarType(), DebugLoc());
+ }
+ Term->setOperand(0, Countable);
+
+ // Remove the combining or.
+ cast<VPInstruction>(Cond)->eraseFromParent();
+
+ return true;
+}
+
/// Returns the VPValue representing the uncountable exit comparison used by
/// AnyOf if the recipes it depends on can be traced back to live-ins and
/// the addresses (in GEP/PtrAdd form) of any (non-masked) load used in
@@ -3102,10 +3217,10 @@ getRecipesForUncountableExit(SmallVectorImpl<VPInstruction *> &Recipes,
return nullptr;
Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
Recipes.push_back(cast<VPInstruction>(GepR));
- } else if (match(V, m_Freeze(m_VPValue(
- Op1, m_VPInstruction<VPInstruction::MaskedCond>(
- m_VPValue(Op2)))))) {
- Worklist.push_back(Op2);
+ } else if (match(V, m_Freeze(m_CombineOr(m_VPInstruction<VPInstruction::MaskedCond>(
+ m_VPValue(Op1)),
+ m_VPValue(Op1))))) {
+ Worklist.push_back(Op1);
Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
Recipes.push_back(cast<VPInstruction>(Op1->getDefiningRecipe()));
} else
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 171fc3b92fe270..56f523b916b7e3 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -390,6 +390,12 @@ struct VPlanTransforms {
DominatorTree &DT,
AssumptionCache *AC);
+ /// If a single exit has multiple conditions combined together, split them
+ /// and create new exiting blocks. Currently limited to a single exit in the
+ /// latch block.
+ static bool splitCombinedExits(VPlan &Plan, PredicatedScalarEvolution &PSE,
+ Loop *TheLoop);
+
/// Update \p Plan to account for uncountable early exits by introducing
/// appropriate branching logic in the latch that handles early exits and the
/// latch exit condition. Multiple exits are handled with a dispatch block
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll b/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll
index fdc571c281c02f..4dbae52ad69209 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll
@@ -273,6 +273,77 @@ exit:
}
define void @combined_exit_conditions(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
+; CHECK-LABEL: VPlan for loop in 'combined_exit_conditions'
+; CHECK: VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT: Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT: Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT: Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT: Live-in ir<20> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<entry>:
+; CHECK-NEXT: Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.ph:
+; CHECK-NEXT: Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION nuw nsw ir<0>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT: CLONE ir<%ee.ptr> = getelementptr inbounds nuw ir<%pred>, vp<[[VP4]]>
+; CHECK-NEXT: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds nuw i32, ir<%ee.ptr>, ir<1>
+; CHECK-NEXT: WIDEN ir<%ee.val> = load vp<[[VP5]]>
+; CHECK-NEXT: WIDEN ir<%ee.cmp> = icmp ne ir<%ee.val>, ir<0>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = freeze ir<%ee.cmp>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = first-active-lane vp<[[VP6]]>
+; CHECK-NEXT: EMIT vp<%uncountable.exit.mask> = active lane mask ir<0>, vp<[[VP7]]>
+; CHECK-NEXT: CLONE ir<%src.ptr> = getelementptr ir<%src>, vp<[[VP4]]>
+; CHECK-NEXT: vp<[[VP8:%[0-9]+]]> = vector-pointer i32, ir<%src.ptr>, ir<1>
+; CHECK-NEXT: WIDEN ir<%data> = load vp<[[VP8]]>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: WIDEN ir<%add> = add nsw ir<%data>, ir<1>
+; CHECK-NEXT: CLONE ir<%dst.ptr> = getelementptr ir<%dst>, vp<[[VP4]]>
+; CHECK-NEXT: vp<[[VP9:%[0-9]+]]> = vector-pointer i32, ir<%dst.ptr>, ir<1>
+; CHECK-NEXT: WIDEN store vp<[[VP9]]>, ir<%add>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = any-of vp<[[VP6]]>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT: EMIT branch-on-two-conds vp<[[VP10]]>, vp<[[VP11]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT: middle.block:
+; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = extract-lane ir<0>, ir<%iv>
+; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = add vp<[[VP13]]>, vp<[[VP7]]>
+; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = icmp eq vp<[[VP14]]>, ir<20>
+; CHECK-NEXT: EMIT branch-on-cond vp<[[VP15]]>
+; CHECK-NEXT: Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<exit>:
+; CHECK-NEXT: No successors
+; CHECK-EMPTY:
+; CHECK-NEXT: scalar.ph:
+; CHECK-NEXT: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP14]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT: Successor(s): ir-bb<for.body>
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<for.body>:
+; CHECK-NEXT: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT: IR %src.ptr = getelementptr inbounds nuw [4 x i8], ptr %src, i64 %iv
+; CHECK-NEXT: IR %data = load i32, ptr %src.ptr, align 4
+; CHECK-NEXT: IR %add = add nsw i32 %data, 1
+; CHECK-NEXT: IR %dst.ptr = getelementptr inbounds nuw [4 x i8], ptr %dst, i64 %iv
+; CHECK-NEXT: IR store i32 %add, ptr %dst.ptr, align 4
+; CHECK-NEXT: IR %ee.ptr = getelementptr inbounds nuw [4 x i8], ptr %pred, i64 %iv
+; CHECK-NEXT: IR %ee.val = load i32, ptr %ee.ptr, align 4
+; CHECK-NEXT: IR %ee.cmp = icmp ne i32 %ee.val, 0
+; CHECK-NEXT: IR %iv.next = add nuw nsw i64 %iv, 1
+; CHECK-NEXT: IR %counted.cmp = icmp eq i64 %iv.next, 20
+; CHECK-NEXT: IR %combined.cond = select i1 %ee.cmp, i1 true, i1 %counted.cmp
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
index 5fd844e186e44e..40ae56a8ec7334 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
@@ -19,6 +19,7 @@
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::replaceSymbolicStrides at 2
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::finalizeSCEVPredicates
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::addMiddleCheck
+; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::splitCombinedExits
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::handleCountableEarlyExits
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::createLoopRegions
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::introduceMasksAndLinearize
diff --git a/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-missed.ll b/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-missed.ll
index c389765dff2d1a..78fa7e1c3d4e70 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-missed.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/vectorization-remarks-missed.ll
@@ -48,6 +48,15 @@
; YAML: --- !Analysis
; YAML-NEXT: Pass: loop-vectorize
+; YAML-NEXT: Name: NoCountableConditionInLatchBlock
+; YAML-NEXT: DebugLoc: { File: source.cpp, Line: 5, Column: 9 }
+; YAML-NEXT: Function: _Z4testPii
+; YAML-NEXT: Args:
+; YAML-NEXT: - String: 'loop not vectorized: '
+; YAML-NEXT: - String: Latch block does not have a countable exit condition
+; YAML-NEXT: ...
+; YAML-NEXT: --- !Analysis
+; YAML-NEXT: Pass: loop-vectorize
; YAML-NEXT: Name: UnsupportedUncountableLoop
; YAML-NEXT: DebugLoc: { File: source.cpp, Line: 5, Column: 9 }
; YAML-NEXT: Function: _Z4testPii
@@ -137,6 +146,15 @@
; YAML-NEXT: ...
; YAML-NEXT: --- !Analysis
; YAML-NEXT: Pass: loop-vectorize
+; YAML-NEXT: Name: RecurrencesInUncountableExitLoop
+; YAML-NEXT: DebugLoc: { File: source.cpp, Line: 27, Column: 3 }
+; YAML-NEXT: Function: test_multiple_failures
+; YAML-NEXT: Args:
+; YAML-NEXT: - String: 'loop not vectorized: '
+; YAML-NEXT: - String: Cannot vectorize uncountable exit loop with reductions or recurrences
+; YAML-NEXT: ...
+; YAML-NEXT: --- !Analysis
+; YAML-NEXT: Pass: loop-vectorize
; YAML-NEXT: Name: UnsupportedUncountableLoop
; YAML-NEXT: DebugLoc: { File: source.cpp, Line: 27, Column: 3 }
; YAML-NEXT: Function: test_multiple_failures
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll b/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll
index f8f561e242bdcb..ef803f993b06db 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll
@@ -4,10 +4,37 @@
define void @combined_exit_conditions(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
; CHECK-LABEL: define void @combined_exit_conditions(
; CHECK-SAME: ptr readonly align 4 dereferenceable(80) [[SRC:%.*]], ptr noalias align 4 dereferenceable(80) [[DST:%.*]], ptr readonly align 4 dereferenceable(80) [[PRED:%.*]]) {
-; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[FOR_BODY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = icmp ne <4 x i32> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT: [[TMP12:%.*]] = freeze <4 x i1> [[TMP2]]
+; CHECK-NEXT: [[TMP3:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP12]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP3]])
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr [4 x i8], ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[TMP0]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i32> poison)
+; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i32> [[WIDE_MASKED_LOAD]], splat (i32 1)
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr [4 x i8], ptr [[DST]], i64 [[INDEX]]
+; CHECK-NEXT: call void @llvm.masked.store.v4i32.p0(<4 x i32> [[TMP4]], ptr align 4 [[TMP5]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP6:%.*]] = freeze <4 x i1> [[TMP12]]
+; CHECK-NEXT: [[TMP7:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP6]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP9:%.*]] = or i1 [[TMP7]], [[TMP8]]
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[INDEX]], [[TMP3]]
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[TMP10]], 20
+; CHECK-NEXT: br i1 [[TMP11]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
; CHECK: [[FOR_BODY1]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY1]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP10]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY1]] ]
; CHECK-NEXT: [[SRC_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[IV]]
; CHECK-NEXT: [[DATA:%.*]] = load i32, ptr [[SRC_PTR]], align 4
; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[DATA]], 1
@@ -19,7 +46,7 @@ define void @combined_exit_conditions(ptr align 4 dereferenceable(80) readonly %
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[COUNTED_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], 20
; CHECK-NEXT: [[COMBINED_COND:%.*]] = select i1 [[EE_CMP]], i1 true, i1 [[COUNTED_CMP]]
-; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT:.*]], label %[[FOR_BODY1]]
+; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -48,10 +75,37 @@ exit:
define void @combined_exit_conditions_swap_comparison_order(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
; CHECK-LABEL: define void @combined_exit_conditions_swap_comparison_order(
; CHECK-SAME: ptr readonly align 4 dereferenceable(80) [[SRC:%.*]], ptr noalias align 4 dereferenceable(80) [[DST:%.*]], ptr readonly align 4 dereferenceable(80) [[PRED:%.*]]) {
-; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[FOR_BODY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <4 x i32> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT: [[TMP12:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP12]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP2]])
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr [4 x i8], ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i32> poison)
+; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i32> [[WIDE_MASKED_LOAD]], splat (i32 1)
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr [4 x i8], ptr [[DST]], i64 [[INDEX]]
+; CHECK-NEXT: call void @llvm.masked.store.v4i32.p0(<4 x i32> [[TMP4]], ptr align 4 [[TMP5]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP6:%.*]] = freeze <4 x i1> [[TMP12]]
+; CHECK-NEXT: [[TMP7:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP6]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP9:%.*]] = or i1 [[TMP7]], [[TMP8]]
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[TMP10]], 20
+; CHECK-NEXT: br i1 [[TMP11]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP10]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY1]] ]
; CHECK-NEXT: [[SRC_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[IV]]
; CHECK-NEXT: [[DATA:%.*]] = load i32, ptr [[SRC_PTR]], align 4
; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[DATA]], 1
@@ -63,7 +117,7 @@ define void @combined_exit_conditions_swap_comparison_order(ptr align 4 derefere
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[COUNTED_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], 20
; CHECK-NEXT: [[COMBINED_COND:%.*]] = select i1 [[COUNTED_CMP]], i1 true, i1 [[EE_CMP]]
-; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT:.*]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -322,10 +376,36 @@ exit:
define void @combined_exit_conditions_binary_or(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
; CHECK-LABEL: define void @combined_exit_conditions_binary_or(
; CHECK-SAME: ptr readonly align 4 dereferenceable(80) [[SRC:%.*]], ptr noalias align 4 dereferenceable(80) [[DST:%.*]], ptr readonly align 4 dereferenceable(80) [[PRED:%.*]]) {
-; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[FOR_BODY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <4 x i32> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP1]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP2]])
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr [4 x i8], ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i32> poison)
+; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i32> [[WIDE_MASKED_LOAD]], splat (i32 1)
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr [4 x i8], ptr [[DST]], i64 [[INDEX]]
+; CHECK-NEXT: call void @llvm.masked.store.v4i32.p0(<4 x i32> [[TMP4]], ptr align 4 [[TMP5]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP6:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP7:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP6]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP9:%.*]] = or i1 [[TMP7]], [[TMP8]]
+; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP10:%.*]] = add i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[TMP10]], 20
+; CHECK-NEXT: br i1 [[TMP11]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP10]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY1]] ]
; CHECK-NEXT: [[SRC_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[IV]]
; CHECK-NEXT: [[DATA:%.*]] = load i32, ptr [[SRC_PTR]], align 4
; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[DATA]], 1
@@ -337,7 +417,7 @@ define void @combined_exit_conditions_binary_or(ptr align 4 dereferenceable(80)
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[COUNTED_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], 20
; CHECK-NEXT: [[COMBINED_COND:%.*]] = or i1 [[EE_CMP]], [[COUNTED_CMP]]
-; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT:.*]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP7:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -406,3 +486,276 @@ for.body:
exit:
ret void
}
+
+define void @combined_exit_conditions_in_latch_with_extra_exit(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred.a, ptr align 4 dereferenceable(80) readonly %pred.b) {
+; CHECK-LABEL: define void @combined_exit_conditions_in_latch_with_extra_exit(
+; CHECK-SAME: ptr readonly align 4 dereferenceable(80) [[SRC:%.*]], ptr noalias align 4 dereferenceable(80) [[DST:%.*]], ptr readonly align 4 dereferenceable(80) [[PRED_A:%.*]], ptr readonly align 4 dereferenceable(80) [[PRED_B:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY_CONT:.*]] ]
+; CHECK-NEXT: [[SRC_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i32, ptr [[SRC_PTR]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[DATA]], 1
+; CHECK-NEXT: [[PRED_A_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED_A]], i64 [[IV]]
+; CHECK-NEXT: [[PRED_A_VAL:%.*]] = load i32, ptr [[PRED_A_PTR]], align 4
+; CHECK-NEXT: [[PRED_A_CMP:%.*]] = icmp ne i32 [[PRED_A_VAL]], 100
+; CHECK-NEXT: br i1 [[PRED_A_CMP]], label %[[EXIT:.*]], label %[[FOR_BODY_CONT]]
+; CHECK: [[FOR_BODY_CONT]]:
+; CHECK-NEXT: [[DST_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[ADD]], ptr [[DST_PTR]], align 4
+; CHECK-NEXT: [[EE_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED_B]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i32, ptr [[EE_PTR]], align 4
+; CHECK-NEXT: [[EE_CMP:%.*]] = icmp ne i32 [[EE_VAL]], 0
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: [[COMBINED_COND:%.*]] = select i1 [[EE_CMP]], i1 true, i1 [[COUNTED_CMP]]
+; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body.cont ]
+ %src.ptr = getelementptr inbounds nuw [4 x i8], ptr %src, i64 %iv
+ %data = load i32, ptr %src.ptr, align 4
+ %add = add nsw i32 %data, 1
+ %pred.a.ptr = getelementptr inbounds nuw [4 x i8], ptr %pred.a, i64 %iv
+ %pred.a.val = load i32, ptr %pred.a.ptr, align 4
+ %pred.a.cmp = icmp ne i32 %pred.a.val, 100
+ br i1 %pred.a.cmp, label %exit, label %for.body.cont
+
+for.body.cont:
+ %dst.ptr = getelementptr inbounds nuw [4 x i8], ptr %dst, i64 %iv
+ store i32 %add, ptr %dst.ptr, align 4
+ %ee.ptr = getelementptr inbounds nuw [4 x i8], ptr %pred.b, i64 %iv
+ %ee.val = load i32, ptr %ee.ptr, align 4
+ %ee.cmp = icmp ne i32 %ee.val, 0
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cmp = icmp eq i64 %iv.next, 20
+ %combined.cond = select i1 %ee.cmp, i1 true, i1 %counted.cmp
+ br i1 %combined.cond, label %exit, label %for.body
+
+exit:
+ ret void
+}
+
+define void @combined_exit_conditions_swap_exit_path(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
+; CHECK-LABEL: define void @combined_exit_conditions_swap_exit_path(
+; CHECK-SAME: ptr readonly align 4 dereferenceable(80) [[SRC:%.*]], ptr noalias align 4 dereferenceable(80) [[DST:%.*]], ptr readonly align 4 dereferenceable(80) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[SRC_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i32, ptr [[SRC_PTR]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[DATA]], 1
+; CHECK-NEXT: [[DST_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[ADD]], ptr [[DST_PTR]], align 4
+; CHECK-NEXT: [[EE_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i32, ptr [[EE_PTR]], align 4
+; CHECK-NEXT: [[EE_CMP:%.*]] = icmp eq i32 [[EE_VAL]], 0
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_CMP:%.*]] = icmp ult i64 [[IV_NEXT]], 20
+; CHECK-NEXT: [[COMBINED_COND:%.*]] = select i1 [[EE_CMP]], i1 true, i1 [[COUNTED_CMP]]
+; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[FOR_BODY]], label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %src.ptr = getelementptr inbounds nuw [4 x i8], ptr %src, i64 %iv
+ %data = load i32, ptr %src.ptr, align 4
+ %add = add nsw i32 %data, 1
+ %dst.ptr = getelementptr inbounds nuw [4 x i8], ptr %dst, i64 %iv
+ store i32 %add, ptr %dst.ptr, align 4
+ %ee.ptr = getelementptr inbounds nuw [4 x i8], ptr %pred, i64 %iv
+ %ee.val = load i32, ptr %ee.ptr, align 4
+ %ee.cmp = icmp eq i32 %ee.val, 0
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cmp = icmp ult i64 %iv.next, 20
+ %combined.cond = select i1 %ee.cmp, i1 true, i1 %counted.cmp
+ br i1 %combined.cond, label %for.body, label %exit
+
+exit:
+ ret void
+}
+
+define i64 @combined_exit_conditions_liveout_iv(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
+; CHECK-LABEL: define i64 @combined_exit_conditions_liveout_iv(
+; CHECK-SAME: ptr readonly align 4 dereferenceable(80) [[SRC:%.*]], ptr noalias align 4 dereferenceable(80) [[DST:%.*]], ptr readonly align 4 dereferenceable(80) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[SRC_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i32, ptr [[SRC_PTR]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[DATA]], 1
+; CHECK-NEXT: [[DST_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[ADD]], ptr [[DST_PTR]], align 4
+; CHECK-NEXT: [[EE_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i32, ptr [[EE_PTR]], align 4
+; CHECK-NEXT: [[EE_CMP:%.*]] = icmp ne i32 [[EE_VAL]], 0
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: [[COMBINED_COND:%.*]] = select i1 [[EE_CMP]], i1 true, i1 [[COUNTED_CMP]]
+; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT:.*]], label %[[FOR_BODY]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[IV_LCSSA:%.*]] = phi i64 [ [[IV]], %[[FOR_BODY]] ]
+; CHECK-NEXT: ret i64 [[IV_LCSSA]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %src.ptr = getelementptr inbounds nuw [4 x i8], ptr %src, i64 %iv
+ %data = load i32, ptr %src.ptr, align 4
+ %add = add nsw i32 %data, 1
+ %dst.ptr = getelementptr inbounds nuw [4 x i8], ptr %dst, i64 %iv
+ store i32 %add, ptr %dst.ptr, align 4
+ %ee.ptr = getelementptr inbounds nuw [4 x i8], ptr %pred, i64 %iv
+ %ee.val = load i32, ptr %ee.ptr, align 4
+ %ee.cmp = icmp ne i32 %ee.val, 0
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cmp = icmp eq i64 %iv.next, 20
+ %combined.cond = select i1 %ee.cmp, i1 true, i1 %counted.cmp
+ br i1 %combined.cond, label %exit, label %for.body
+
+exit:
+ ret i64 %iv
+}
+
+define i64 @combined_exit_conditions_liveout_iv_next(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
+; CHECK-LABEL: define i64 @combined_exit_conditions_liveout_iv_next(
+; CHECK-SAME: ptr readonly align 4 dereferenceable(80) [[SRC:%.*]], ptr noalias align 4 dereferenceable(80) [[DST:%.*]], ptr readonly align 4 dereferenceable(80) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[SRC_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i32, ptr [[SRC_PTR]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[DATA]], 1
+; CHECK-NEXT: [[DST_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[ADD]], ptr [[DST_PTR]], align 4
+; CHECK-NEXT: [[EE_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i32, ptr [[EE_PTR]], align 4
+; CHECK-NEXT: [[EE_CMP:%.*]] = icmp ne i32 [[EE_VAL]], 0
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: [[COMBINED_COND:%.*]] = select i1 [[EE_CMP]], i1 true, i1 [[COUNTED_CMP]]
+; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT:.*]], label %[[FOR_BODY]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[IV_NEXT_LCSSA:%.*]] = phi i64 [ [[IV_NEXT]], %[[FOR_BODY]] ]
+; CHECK-NEXT: ret i64 [[IV_NEXT_LCSSA]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %src.ptr = getelementptr inbounds nuw [4 x i8], ptr %src, i64 %iv
+ %data = load i32, ptr %src.ptr, align 4
+ %add = add nsw i32 %data, 1
+ %dst.ptr = getelementptr inbounds nuw [4 x i8], ptr %dst, i64 %iv
+ store i32 %add, ptr %dst.ptr, align 4
+ %ee.ptr = getelementptr inbounds nuw [4 x i8], ptr %pred, i64 %iv
+ %ee.val = load i32, ptr %ee.ptr, align 4
+ %ee.cmp = icmp ne i32 %ee.val, 0
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cmp = icmp eq i64 %iv.next, 20
+ %combined.cond = select i1 %ee.cmp, i1 true, i1 %counted.cmp
+ br i1 %combined.cond, label %exit, label %for.body
+
+exit:
+ ret i64 %iv.next
+}
+
+define i32 @combined_exit_conditions_liveout_ee_val(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
+; CHECK-LABEL: define i32 @combined_exit_conditions_liveout_ee_val(
+; CHECK-SAME: ptr readonly align 4 dereferenceable(80) [[SRC:%.*]], ptr noalias align 4 dereferenceable(80) [[DST:%.*]], ptr readonly align 4 dereferenceable(80) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[SRC_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i32, ptr [[SRC_PTR]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[DATA]], 1
+; CHECK-NEXT: [[DST_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[ADD]], ptr [[DST_PTR]], align 4
+; CHECK-NEXT: [[EE_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i32, ptr [[EE_PTR]], align 4
+; CHECK-NEXT: [[EE_CMP:%.*]] = icmp ugt i32 [[EE_VAL]], 500
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: [[COMBINED_COND:%.*]] = select i1 [[EE_CMP]], i1 true, i1 [[COUNTED_CMP]]
+; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT:.*]], label %[[FOR_BODY]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[EE_VAL_LCSSA:%.*]] = phi i32 [ [[EE_VAL]], %[[FOR_BODY]] ]
+; CHECK-NEXT: ret i32 [[EE_VAL_LCSSA]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %src.ptr = getelementptr inbounds nuw [4 x i8], ptr %src, i64 %iv
+ %data = load i32, ptr %src.ptr, align 4
+ %add = add nsw i32 %data, 1
+ %dst.ptr = getelementptr inbounds nuw [4 x i8], ptr %dst, i64 %iv
+ store i32 %add, ptr %dst.ptr, align 4
+ %ee.ptr = getelementptr inbounds nuw [4 x i8], ptr %pred, i64 %iv
+ %ee.val = load i32, ptr %ee.ptr, align 4
+ %ee.cmp = icmp ugt i32 %ee.val, 500
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cmp = icmp eq i64 %iv.next, 20
+ %combined.cond = select i1 %ee.cmp, i1 true, i1 %counted.cmp
+ br i1 %combined.cond, label %exit, label %for.body
+
+exit:
+ ret i32 %ee.val
+}
+
+;; Caused a crash when vectorizing due to trying to tail fold.
+define void @short_trip_count_no_tf(ptr noalias align 4 dereferenceable(80) %dst, ptr noalias align 4 dereferenceable(80) readonly %pred) {
+; CHECK-LABEL: define void @short_trip_count_no_tf(
+; CHECK-SAME: ptr noalias align 4 dereferenceable(80) [[DST:%.*]], ptr noalias readonly align 4 dereferenceable(80) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[DST_PTR:%.*]] = getelementptr inbounds nuw i32, ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT: store i32 1, ptr [[DST_PTR]], align 4
+; CHECK-NEXT: [[EE_PTR:%.*]] = getelementptr inbounds nuw i32, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i32, ptr [[EE_PTR]], align 4
+; CHECK-NEXT: [[EE_CMP:%.*]] = icmp ne i32 [[EE_VAL]], 0
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], 7
+; CHECK-NEXT: [[OR:%.*]] = or i1 [[EE_CMP]], [[COUNTED_CMP]]
+; CHECK-NEXT: br i1 [[OR]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+ %dst.ptr = getelementptr inbounds nuw i32, ptr %dst, i64 %iv
+ store i32 1, ptr %dst.ptr, align 4
+ %ee.ptr = getelementptr inbounds nuw i32, ptr %pred, i64 %iv
+ %ee.val = load i32, ptr %ee.ptr, align 4
+ %ee.cmp = icmp ne i32 %ee.val, 0
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cmp = icmp eq i64 %iv.next, 7
+ %or = or i1 %ee.cmp, %counted.cmp
+ br i1 %or, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits_epilogue.ll b/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits_epilogue.ll
new file mode 100644
index 00000000000000..b83534c21a3196
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits_epilogue.ll
@@ -0,0 +1,73 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=8 -epilogue-vectorization-force-VF=2 -force-target-supports-masked-memory-ops -enable-early-exit-vectorization-with-side-effects | FileCheck %s
+
+define void @combined_exit_conditions(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
+; CHECK-LABEL: define void @combined_exit_conditions(
+; CHECK-SAME: ptr readonly align 4 dereferenceable(80) [[SRC:%.*]], ptr noalias align 4 dereferenceable(80) [[DST:%.*]], ptr readonly align 4 dereferenceable(80) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP0]], align 4
+; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <8 x i32> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT: [[TMP2:%.*]] = freeze <8 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP3:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v8i1(<8 x i1> [[TMP2]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 0, i64 [[TMP3]])
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr [4 x i8], ptr [[SRC]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 4 [[TMP4]], <8 x i1> [[UNCOUNTABLE_EXIT_MASK]], <8 x i32> poison)
+; CHECK-NEXT: [[TMP5:%.*]] = add nsw <8 x i32> [[WIDE_MASKED_LOAD]], splat (i32 1)
+; CHECK-NEXT: [[TMP6:%.*]] = getelementptr [4 x i8], ptr [[DST]], i64 [[INDEX]]
+; CHECK-NEXT: call void @llvm.masked.store.v8i32.p0(<8 x i32> [[TMP5]], ptr align 4 [[TMP6]], <8 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP7:%.*]] = freeze <8 x i1> [[TMP2]]
+; CHECK-NEXT: [[TMP8:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP7]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
+; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 16
+; CHECK-NEXT: [[TMP10:%.*]] = or i1 [[TMP8]], [[TMP9]]
+; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[INDEX]], [[TMP3]]
+; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[TMP11]], 20
+; CHECK-NEXT: br i1 [[TMP12]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP11]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[SRC_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i32, ptr [[SRC_PTR]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[DATA]], 1
+; CHECK-NEXT: [[DST_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[ADD]], ptr [[DST_PTR]], align 4
+; CHECK-NEXT: [[EE_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i32, ptr [[EE_PTR]], align 4
+; CHECK-NEXT: [[EE_CMP:%.*]] = icmp ne i32 [[EE_VAL]], 0
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: [[COMBINED_COND:%.*]] = select i1 [[EE_CMP]], i1 true, i1 [[COUNTED_CMP]]
+; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %src.ptr = getelementptr inbounds nuw [4 x i8], ptr %src, i64 %iv
+ %data = load i32, ptr %src.ptr, align 4
+ %add = add nsw i32 %data, 1
+ %dst.ptr = getelementptr inbounds nuw [4 x i8], ptr %dst, i64 %iv
+ store i32 %add, ptr %dst.ptr, align 4
+ %ee.ptr = getelementptr inbounds nuw [4 x i8], ptr %pred, i64 %iv
+ %ee.val = load i32, ptr %ee.ptr, align 4
+ %ee.cmp = icmp ne i32 %ee.val, 0
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cmp = icmp eq i64 %iv.next, 20
+ %combined.cond = select i1 %ee.cmp, i1 true, i1 %counted.cmp
+ br i1 %combined.cond, label %exit, label %for.body
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll b/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll
index 71650a4c365673..3a0fab8118424c 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll
@@ -1049,10 +1049,9 @@ invalid.block:
unreachable
}
-define void @combined_exit_conditions(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) !dbg !76 {
-; CHECK-DEBUG-LABEL: LV: Checking a loop in 'combined_exit_conditions'
-; CHECK-DEBUG: LV: Not vectorizing: Cannot vectorize uncountable loop.
-; CHECK-REMARK: foo.c:340:3: loop not vectorized: Cannot vectorize uncountable loop
+define void @combined_exit_conditions(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
+; CHECK-LABEL: LV: Checking a loop in 'combined_exit_conditions'
+; CHECK: LV: We can vectorize this loop!
entry:
br label %for.body, !dbg !77
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll b/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll
index bdd795ac02cf4a..ec93f246e9f93f 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll
@@ -1085,6 +1085,77 @@ exit:
ret i16 %data
}
+define i64 @uncountable_exit_with_live_out_iv(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
+; CHECK-LABEL: define i64 @uncountable_exit_with_live_out_iv(
+; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP0]], align 2
+; CHECK-NEXT: [[TMP1:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP1]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP2]])
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x i16> @llvm.masked.load.v4i16.p0(ptr align 2 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i16> poison)
+; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.v4i16.p0(<4 x i16> [[TMP4]], ptr align 2 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP5:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], [[TMP2]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[TMP9]], 20
+; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP9]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
+; CHECK: [[FOR_INC]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[IV_LCSSA:%.*]] = phi i64 [ [[IV]], %[[FOR_INC]] ], [ [[IV]], %[[FOR_BODY]] ], [ 19, %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i64 [[IV_LCSSA]]
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.inc ]
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %for.inc
+
+for.inc:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %for.body
+
+exit:
+ ret i64 %iv
+}
+
define void @uncountable_exit_with_constant_nonunit_stride(ptr dereferenceable(4000) noalias %array, ptr align 2 dereferenceable(4000) readonly %pred) {
; CHECK-LABEL: define void @uncountable_exit_with_constant_nonunit_stride(
; CHECK-SAME: ptr noalias dereferenceable(4000) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(4000) [[PRED:%.*]]) {
@@ -1198,7 +1269,7 @@ define i32 @uncountable_exit_with_separate_exit_block(ptr dereferenceable(40) no
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[TMP3]], 4
; CHECK-NEXT: [[TMP42:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
; CHECK-NEXT: [[TMP43:%.*]] = or i1 [[TMP41]], [[TMP42]]
-; CHECK-NEXT: br i1 [[TMP43]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP43]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[TMP44:%.*]] = add i64 [[TMP3]], [[TMP14]]
; CHECK-NEXT: [[TMP45:%.*]] = icmp eq i64 [[TMP44]], 20
@@ -1218,7 +1289,7 @@ define i32 @uncountable_exit_with_separate_exit_block(ptr dereferenceable(40) no
; CHECK: [[FOR_INC]]:
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT_COUNTABLE]], label %[[FOR_BODY1]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT_COUNTABLE]], label %[[FOR_BODY1]], !llvm.loop [[LOOP13:![0-9]+]]
; CHECK: [[EXIT_COUNTABLE]]:
; CHECK-NEXT: ret i32 0
; CHECK: [[EXIT_UNCOUNTABLE]]:
@@ -1341,7 +1412,7 @@ define void @swapped_cmp_operands(ptr dereferenceable(40) noalias %array, ptr al
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
; CHECK-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
-; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP12:![0-9]+]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP14:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], [[TMP3]]
; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[TMP9]], 20
@@ -1361,7 +1432,7 @@ define void @swapped_cmp_operands(ptr dereferenceable(40) noalias %array, ptr al
; CHECK: [[FOR_INC]]:
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
diff --git a/llvm/test/Transforms/LoopVectorize/uncountable-single-exit-loops.ll b/llvm/test/Transforms/LoopVectorize/uncountable-single-exit-loops.ll
index e7ecf459ca6c93..327c9f1668854b 100644
--- a/llvm/test/Transforms/LoopVectorize/uncountable-single-exit-loops.ll
+++ b/llvm/test/Transforms/LoopVectorize/uncountable-single-exit-loops.ll
@@ -4,12 +4,14 @@
; CHECK-LABEL: LV: Checking a loop in 'latch_exit_cannot_compute_btc_due_to_step'
; CHECK: LV: Did not find one integer induction var.
+; CHECK-NEXT: LV: Not vectorizing: Cannot determine exact exit count for latch block.
; CHECK-NEXT: LV: Not vectorizing: Cannot vectorize uncountable loop.
; CHECK-NEXT: LV: Not vectorizing: Cannot prove legality.
; CHECK-LABEL: LV: Checking a loop in 'header_exit_cannot_compute_btc_due_to_step'
; CHECK: LV: Found an induction variable.
; CHECK-NEXT: LV: Did not find one integer induction var.
+; CHECK-NEXT: LV: Not vectorizing: Cannot determine exact exit count for latch block.
; CHECK-NEXT: LV: Not vectorizing: Cannot vectorize uncountable loop.
; CHECK-NEXT: LV: Not vectorizing: Cannot prove legality.
diff --git a/llvm/test/Transforms/PhaseOrdering/AArch64/countable-and-uncountable-exits-combined.ll b/llvm/test/Transforms/PhaseOrdering/AArch64/countable-and-uncountable-exits-combined.ll
index 2b2f0fbc87762f..e7ed25c7954d72 100644
--- a/llvm/test/Transforms/PhaseOrdering/AArch64/countable-and-uncountable-exits-combined.ll
+++ b/llvm/test/Transforms/PhaseOrdering/AArch64/countable-and-uncountable-exits-combined.ll
@@ -1,5 +1,5 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -passes="default<O3>" -S %s | FileCheck %s
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes="default<O3>" -enable-early-exit-vectorization-with-side-effects -S %s | FileCheck %s
target triple = "aarch64"
@@ -26,9 +26,41 @@ define void @foo() #0 {
; CHECK-LABEL: define void @foo(
; CHECK-SAME: ) local_unnamed_addr #[[ATTR0:[0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = tail call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 2
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ugt i64 [[TMP0]], 2500
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[FOR_BODY_PREHEADER:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 10000, [[TMP1]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub nuw nsw i64 10000, [[N_MOD_VF]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[INDVARS_IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds nuw [4 x i8], ptr @c, i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 4 x i32>, ptr [[TMP2]], align 4
+; CHECK-NEXT: [[WIDE_LOAD_FR:%.*]] = freeze <vscale x 4 x i32> [[WIDE_LOAD]]
+; CHECK-NEXT: [[TMP3:%.*]] = icmp ne <vscale x 4 x i32> [[WIDE_LOAD_FR]], zeroinitializer
+; CHECK-NEXT: [[TMP4:%.*]] = tail call i64 @llvm.experimental.cttz.elts.i64.nxv4i1(<vscale x 4 x i1> [[TMP3]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = tail call <vscale x 4 x i1> @llvm.get.active.lane.mask.nxv4i1.i64(i64 0, i64 [[TMP4]])
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr [4 x i8], ptr @src, i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = tail call <vscale x 4 x i32> @llvm.masked.load.nxv4i32.p0(ptr align 4 [[TMP5]], <vscale x 4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <vscale x 4 x i32> poison)
+; CHECK-NEXT: [[TMP6:%.*]] = add nsw <vscale x 4 x i32> [[WIDE_MASKED_LOAD]], splat (i32 42)
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr [4 x i8], ptr @dst, i64 [[INDEX]]
+; CHECK-NEXT: tail call void @llvm.masked.store.nxv4i32.p0(<vscale x 4 x i32> [[TMP6]], ptr align 4 [[TMP7]], <vscale x 4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP8:%.*]] = tail call i1 @llvm.vector.reduce.or.nxv4i1(<vscale x 4 x i1> [[TMP3]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], [[TMP1]]
+; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: [[TMP10:%.*]] = or i1 [[TMP8]], [[TMP9]]
+; CHECK-NEXT: br i1 [[TMP10]], label %[[MIDDLE_BLOCK:.*]], label %[[FOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP11:%.*]] = add i64 [[INDEX]], [[TMP4]]
+; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[TMP11]], 10000
+; CHECK-NEXT: br i1 [[TMP12]], label %[[EXIT:.*]], label %[[FOR_BODY_PREHEADER]]
+; CHECK: [[FOR_BODY_PREHEADER]]:
+; CHECK-NEXT: [[INDVARS_IV_PH:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[TMP11]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ [[INDVARS_IV_NEXT:%.*]], %[[FOR_BODY1]] ], [ [[INDVARS_IV_PH]], %[[FOR_BODY_PREHEADER]] ]
; CHECK-NEXT: [[ARRAYIDX:%.*]] = getelementptr inbounds nuw [4 x i8], ptr @src, i64 [[INDVARS_IV]]
; CHECK-NEXT: [[SRC_PTR:%.*]] = load i32, ptr [[ARRAYIDX]], align 4
; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[SRC_PTR]], 42
@@ -40,7 +72,7 @@ define void @foo() #0 {
; CHECK-NEXT: [[INDVARS_IV_NEXT]] = add nuw nsw i64 [[INDVARS_IV]], 1
; CHECK-NEXT: [[EXITCOND_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT]], 10000
; CHECK-NEXT: [[OR_COND:%.*]] = select i1 [[EE_COND]], i1 true, i1 [[EXITCOND_NOT]]
-; CHECK-NEXT: br i1 [[OR_COND]], label %[[EXIT:.*]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[OR_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
>From bb461a0c5e7e8d8a0ddf4ae111e1bee44ee2aecb Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Wed, 23 Sep 2026 12:02:07 +0000
Subject: [PATCH 2/3] Adapt to upstream changes
---
.../Vectorize/LoopVectorizationPlanner.cpp | 3 +-
.../Transforms/Vectorize/VPlanTransforms.cpp | 13 ++++----
.../VPlan/early_exit_with_stores_vplan.ll | 33 ++++++++++---------
.../early_exit_combined_exits.ll | 12 +++----
.../early_exit_combined_exits_epilogue.ll | 4 +--
.../LoopVectorize/early_exit_with_stores.ll | 4 +--
.../LoopVectorize/fold-epilogue-tail.ll | 2 +-
7 files changed, 36 insertions(+), 35 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index 17ea2932ddd50b..fe5d7ee46bed73 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -914,7 +914,8 @@ bool LoopVectorizationPlanner::isCandidateForEpilogueVectorization(
// non-latch exits properly. It may be fine, but it needs auditted and
// tested.
// TODO: Add support for loops with an early exit.
- if (OrigLoop->getExitingBlock() != OrigLoop->getLoopLatch())
+ if (OrigLoop->getExitingBlock() != OrigLoop->getLoopLatch() ||
+ Legal->hasUncountableEarlyExit())
return false;
return true;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 8d61d7e54373d3..1d1431c0bbb2d7 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -3093,8 +3093,7 @@ bool VPlanTransforms::splitCombinedExits(VPlan &Plan,
// Create new terminator for uncountable condition.
VPBuilder EEBuilder(LatchVPBB);
if (NeedsFreeze)
- Uncountable = EEBuilder.createScalarFreeze(
- Uncountable, Uncountable->getScalarType(), DebugLoc());
+ Uncountable = EEBuilder.createFreeze(Uncountable);
EEBuilder.createNaryOp(VPInstruction::BranchOnCond, {Uncountable});
// We need to connect the uncountable exit to the sole exit block. The
@@ -3109,8 +3108,7 @@ bool VPlanTransforms::splitCombinedExits(VPlan &Plan,
// Set condition for latch block to countable condition.
if (NeedsFreeze) {
VPBuilder NewLatchBuilder(Term);
- Countable = NewLatchBuilder.createScalarFreeze(
- Countable, Countable->getScalarType(), DebugLoc());
+ Countable = NewLatchBuilder.createFreeze(Countable);
}
Term->setOperand(0, Countable);
@@ -3217,9 +3215,10 @@ getRecipesForUncountableExit(SmallVectorImpl<VPInstruction *> &Recipes,
return nullptr;
Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
Recipes.push_back(cast<VPInstruction>(GepR));
- } else if (match(V, m_Freeze(m_CombineOr(m_VPInstruction<VPInstruction::MaskedCond>(
- m_VPValue(Op1)),
- m_VPValue(Op1))))) {
+ } else if (match(V, m_Freeze(m_CombineOr(
+ m_VPInstruction<VPInstruction::MaskedCond>(
+ m_VPValue(Op1)),
+ m_VPValue(Op1))))) {
Worklist.push_back(Op1);
Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
Recipes.push_back(cast<VPInstruction>(Op1->getDefiningRecipe()));
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll b/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll
index 4dbae52ad69209..c23eb31dbe0c30 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll
@@ -122,10 +122,10 @@ define void @loop_contains_store_after_uncountable_exit(ptr dereferenceable(40)
; CHECK-NEXT: EMIT vp<%uncountable.exit.mask> = active lane mask ir<0>, vp<[[VP7]]>
; CHECK-NEXT: CLONE ir<%st.addr> = getelementptr ir<%array>, vp<[[VP4]]>
; CHECK-NEXT: vp<[[VP8:%[0-9]+]]> = vector-pointer i16, ir<%st.addr>, ir<1>
-; CHECK-NEXT: WIDEN ir<%data> = load vp<[[VP8]]>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: WIDEN ir<%data> = load vp<[[VP8]]>, vp<%uncountable.exit.mask> (!vplan.execution.frequency 8935141660703064064 (96.88%, estimated))
; CHECK-NEXT: WIDEN ir<%inc> = add nsw ir<%data>, ir<1>
; CHECK-NEXT: vp<[[VP9:%[0-9]+]]> = vector-pointer i16, ir<%st.addr>, ir<1>
-; CHECK-NEXT: WIDEN store vp<[[VP9]]>, ir<%inc>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: WIDEN store vp<[[VP9]]>, ir<%inc>, vp<%uncountable.exit.mask> (!vplan.execution.frequency 8935141660703064064 (96.88%, estimated))
; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = any-of vp<[[VP6]]>
; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
@@ -297,35 +297,36 @@ define void @combined_exit_conditions(ptr align 4 dereferenceable(80) readonly %
; CHECK-NEXT: WIDEN ir<%ee.val> = load vp<[[VP5]]>
; CHECK-NEXT: WIDEN ir<%ee.cmp> = icmp ne ir<%ee.val>, ir<0>
; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = freeze ir<%ee.cmp>
-; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = first-active-lane vp<[[VP6]]>
-; CHECK-NEXT: EMIT vp<%uncountable.exit.mask> = active lane mask ir<0>, vp<[[VP7]]>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = freeze vp<[[VP6]]>
+; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = first-active-lane vp<[[VP7]]>
+; CHECK-NEXT: EMIT vp<%uncountable.exit.mask> = active lane mask ir<0>, vp<[[VP8]]>
; CHECK-NEXT: CLONE ir<%src.ptr> = getelementptr ir<%src>, vp<[[VP4]]>
-; CHECK-NEXT: vp<[[VP8:%[0-9]+]]> = vector-pointer i32, ir<%src.ptr>, ir<1>
-; CHECK-NEXT: WIDEN ir<%data> = load vp<[[VP8]]>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: vp<[[VP9:%[0-9]+]]> = vector-pointer i32, ir<%src.ptr>, ir<1>
+; CHECK-NEXT: WIDEN ir<%data> = load vp<[[VP9]]>, vp<%uncountable.exit.mask>
; CHECK-NEXT: WIDEN ir<%add> = add nsw ir<%data>, ir<1>
; CHECK-NEXT: CLONE ir<%dst.ptr> = getelementptr ir<%dst>, vp<[[VP4]]>
-; CHECK-NEXT: vp<[[VP9:%[0-9]+]]> = vector-pointer i32, ir<%dst.ptr>, ir<1>
-; CHECK-NEXT: WIDEN store vp<[[VP9]]>, ir<%add>, vp<%uncountable.exit.mask>
-; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = any-of vp<[[VP6]]>
+; CHECK-NEXT: vp<[[VP10:%[0-9]+]]> = vector-pointer i32, ir<%dst.ptr>, ir<1>
+; CHECK-NEXT: WIDEN store vp<[[VP10]]>, ir<%add>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = any-of vp<[[VP7]]>
; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
-; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
-; CHECK-NEXT: EMIT branch-on-two-conds vp<[[VP10]]>, vp<[[VP11]]>
+; CHECK-NEXT: EMIT vp<[[VP12:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT: EMIT branch-on-two-conds vp<[[VP11]]>, vp<[[VP12]]>
; CHECK-NEXT: No successors
; CHECK-NEXT: }
; CHECK-NEXT: Successor(s): middle.block, middle.block
; CHECK-EMPTY:
; CHECK-NEXT: middle.block:
-; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = extract-lane ir<0>, ir<%iv>
-; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = add vp<[[VP13]]>, vp<[[VP7]]>
-; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = icmp eq vp<[[VP14]]>, ir<20>
-; CHECK-NEXT: EMIT branch-on-cond vp<[[VP15]]>
+; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = extract-lane ir<0>, ir<%iv>
+; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = add vp<[[VP14]]>, vp<[[VP8]]>
+; CHECK-NEXT: EMIT vp<[[VP16:%[0-9]+]]> = icmp eq vp<[[VP15]]>, ir<20>
+; CHECK-NEXT: EMIT branch-on-cond vp<[[VP16]]>
; CHECK-NEXT: Successor(s): ir-bb<exit>, scalar.ph
; CHECK-EMPTY:
; CHECK-NEXT: ir-bb<exit>:
; CHECK-NEXT: No successors
; CHECK-EMPTY:
; CHECK-NEXT: scalar.ph:
-; CHECK-NEXT: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP14]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP15]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
; CHECK-NEXT: Successor(s): ir-bb<for.body>
; CHECK-EMPTY:
; CHECK-NEXT: ir-bb<for.body>:
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll b/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll
index ef803f993b06db..cb463c98a61819 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll
@@ -14,14 +14,14 @@ define void @combined_exit_conditions(ptr align 4 dereferenceable(80) readonly %
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
; CHECK-NEXT: [[TMP2:%.*]] = icmp ne <4 x i32> [[WIDE_LOAD]], zeroinitializer
; CHECK-NEXT: [[TMP12:%.*]] = freeze <4 x i1> [[TMP2]]
-; CHECK-NEXT: [[TMP3:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP12]], i1 false)
+; CHECK-NEXT: [[TMP6:%.*]] = freeze <4 x i1> [[TMP12]]
+; CHECK-NEXT: [[TMP3:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP6]], i1 false)
; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP3]])
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr [4 x i8], ptr [[SRC]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[TMP0]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i32> poison)
; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i32> [[WIDE_MASKED_LOAD]], splat (i32 1)
; CHECK-NEXT: [[TMP5:%.*]] = getelementptr [4 x i8], ptr [[DST]], i64 [[INDEX]]
; CHECK-NEXT: call void @llvm.masked.store.v4i32.p0(<4 x i32> [[TMP4]], ptr align 4 [[TMP5]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
-; CHECK-NEXT: [[TMP6:%.*]] = freeze <4 x i1> [[TMP12]]
; CHECK-NEXT: [[TMP7:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP6]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
@@ -85,14 +85,14 @@ define void @combined_exit_conditions_swap_comparison_order(ptr align 4 derefere
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <4 x i32> [[WIDE_LOAD]], zeroinitializer
; CHECK-NEXT: [[TMP12:%.*]] = freeze <4 x i1> [[TMP1]]
-; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP12]], i1 false)
+; CHECK-NEXT: [[TMP6:%.*]] = freeze <4 x i1> [[TMP12]]
+; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP6]], i1 false)
; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP2]])
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr [4 x i8], ptr [[SRC]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i32> poison)
; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i32> [[WIDE_MASKED_LOAD]], splat (i32 1)
; CHECK-NEXT: [[TMP5:%.*]] = getelementptr [4 x i8], ptr [[DST]], i64 [[INDEX]]
; CHECK-NEXT: call void @llvm.masked.store.v4i32.p0(<4 x i32> [[TMP4]], ptr align 4 [[TMP5]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
-; CHECK-NEXT: [[TMP6:%.*]] = freeze <4 x i1> [[TMP12]]
; CHECK-NEXT: [[TMP7:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP6]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
@@ -385,14 +385,14 @@ define void @combined_exit_conditions_binary_or(ptr align 4 dereferenceable(80)
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP0]], align 4
; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <4 x i32> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP1]], i1 false)
+; CHECK-NEXT: [[TMP6:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP6]], i1 false)
; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP2]])
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr [4 x i8], ptr [[SRC]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x i32> @llvm.masked.load.v4i32.p0(ptr align 4 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i32> poison)
; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i32> [[WIDE_MASKED_LOAD]], splat (i32 1)
; CHECK-NEXT: [[TMP5:%.*]] = getelementptr [4 x i8], ptr [[DST]], i64 [[INDEX]]
; CHECK-NEXT: call void @llvm.masked.store.v4i32.p0(<4 x i32> [[TMP4]], ptr align 4 [[TMP5]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
-; CHECK-NEXT: [[TMP6:%.*]] = freeze <4 x i1> [[TMP1]]
; CHECK-NEXT: [[TMP7:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP6]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits_epilogue.ll b/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits_epilogue.ll
index b83534c21a3196..c7d7e4637da1f6 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits_epilogue.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits_epilogue.ll
@@ -14,14 +14,14 @@ define void @combined_exit_conditions(ptr align 4 dereferenceable(80) readonly %
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <8 x i32>, ptr [[TMP0]], align 4
; CHECK-NEXT: [[TMP1:%.*]] = icmp ne <8 x i32> [[WIDE_LOAD]], zeroinitializer
; CHECK-NEXT: [[TMP2:%.*]] = freeze <8 x i1> [[TMP1]]
-; CHECK-NEXT: [[TMP3:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v8i1(<8 x i1> [[TMP2]], i1 false)
+; CHECK-NEXT: [[TMP7:%.*]] = freeze <8 x i1> [[TMP2]]
+; CHECK-NEXT: [[TMP3:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v8i1(<8 x i1> [[TMP7]], i1 false)
; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <8 x i1> @llvm.get.active.lane.mask.v8i1.i64(i64 0, i64 [[TMP3]])
; CHECK-NEXT: [[TMP4:%.*]] = getelementptr [4 x i8], ptr [[SRC]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <8 x i32> @llvm.masked.load.v8i32.p0(ptr align 4 [[TMP4]], <8 x i1> [[UNCOUNTABLE_EXIT_MASK]], <8 x i32> poison)
; CHECK-NEXT: [[TMP5:%.*]] = add nsw <8 x i32> [[WIDE_MASKED_LOAD]], splat (i32 1)
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr [4 x i8], ptr [[DST]], i64 [[INDEX]]
; CHECK-NEXT: call void @llvm.masked.store.v8i32.p0(<8 x i32> [[TMP5]], ptr align 4 [[TMP6]], <8 x i1> [[UNCOUNTABLE_EXIT_MASK]])
-; CHECK-NEXT: [[TMP7:%.*]] = freeze <8 x i1> [[TMP2]]
; CHECK-NEXT: [[TMP8:%.*]] = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> [[TMP7]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 16
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll b/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll
index ec93f246e9f93f..f02553b69daa7d 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll
@@ -1097,13 +1097,13 @@ define i64 @uncountable_exit_with_live_out_iv(ptr dereferenceable(40) noalias %a
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP0]], align 2
; CHECK-NEXT: [[TMP1:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
-; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP1]], i1 false)
+; CHECK-NEXT: [[TMP5:%.*]] = freeze <4 x i1> [[TMP1]]
+; CHECK-NEXT: [[TMP2:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP5]], i1 false)
; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP2]])
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[INDEX]]
; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x i16> @llvm.masked.load.v4i16.p0(ptr align 2 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i16> poison)
; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
; CHECK-NEXT: call void @llvm.masked.store.v4i16.p0(<4 x i16> [[TMP4]], ptr align 2 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
-; CHECK-NEXT: [[TMP5:%.*]] = freeze <4 x i1> [[TMP1]]
; CHECK-NEXT: [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
diff --git a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
index 1b3964fe1978de..3a35fa12e7a37c 100644
--- a/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
+++ b/llvm/test/Transforms/LoopVectorize/fold-epilogue-tail.ll
@@ -196,7 +196,7 @@ exit:
; the check line should be changed to: Epilogue tail-folding is not supported yet for early-exit loops, same as the case above.
define void @combined_exit_conditions(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
; CHECK-DISABLED-EARLY-EXIT-LABEL: LV: Checking a loop in 'combined_exit_conditions'
-; CHECK-DISABLED-EARLY-EXIT: remark: <unknown>:0:0: loop not vectorized: Cannot vectorize uncountable loop
+; CHECK-DISABLED-EARLY-EXIT: remark: <unknown>:0:0: Epilogue tail-folding is not supported yet for early-exit loops
;
entry:
br label %for.body
>From f156d88e7896ae8e10a7079f527faff04293960c Mon Sep 17 00:00:00 2001
From: Graham Hunter <graham.hunter at arm.com>
Date: Wed, 23 Sep 2026 12:35:23 +0000
Subject: [PATCH 3/3] Address comments
---
.../Vectorize/LoopVectorizationLegality.cpp | 2 +-
.../Transforms/Vectorize/LoopVectorize.cpp | 8 ++--
.../VPlan/vplan-print-before-after-all.ll | 1 -
.../early_exit_combined_exits.ll | 44 +++++++++++++++++++
.../LoopVectorize/early_exit_legality.ll | 16 +++----
.../early_exit_store_legality.ll | 2 +-
.../uncountable-single-exit-loops.ll | 4 +-
7 files changed, 60 insertions(+), 17 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index d9a93387d50148..325bfaf10c1ca6 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -1767,7 +1767,7 @@ bool LoopVectorizationLegality::isVectorizableEarlyExitLoop() {
if (isa<SCEVCouldNotCompute>(PSE.getSE()->getPredicatedExitCount(
TheLoop, LatchBB, &Predicates, ScalarEvolution::SymbolicMaximum))) {
reportVectorizationFailure(
- "Cannot determine exact exit count for latch block",
+ "Cannot determine symbolic max exit count for latch block",
"Cannot vectorize early exit loop",
"UnknownLatchExitCountEarlyExitLoop", ORE, TheLoop);
return false;
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 43aceff4a7b8e7..708548463f5cca 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -6503,16 +6503,16 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
RUN_VPLAN_PASS(VPlanTransforms::addMiddleCheck, *VPlan0);
- if (!RUN_VPLAN_PASS(VPlanTransforms::splitCombinedExits, *VPlan0, PSE,
- OrigLoop))
- return nullptr;
-
// If we're vectorizing a loop with an uncountable exit, make sure that the
// recipes are safe to handle.
// TODO: Remove this once we can properly check the VPlan itself for both
// the presence of an uncountable exit and the presence of stores in
// the loop inside handleUncountableEarlyExits itself.
if (Legal->hasUncountableEarlyExit()) {
+ if (!RUN_VPLAN_PASS(VPlanTransforms::splitCombinedExits, *VPlan0, PSE,
+ OrigLoop))
+ return nullptr;
+
// TODO: Check target preference for style.
UncountableExitStyle EEStyle =
Legal->hasUncountableExitWithSideEffects()
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
index 40ae56a8ec7334..5fd844e186e44e 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/vplan-print-before-after-all.ll
@@ -19,7 +19,6 @@
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::replaceSymbolicStrides at 2
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::finalizeSCEVPredicates
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::addMiddleCheck
-; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::splitCombinedExits
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::handleCountableEarlyExits
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::createLoopRegions
; CHECK: VPlan for loop in 'foo' [[BEFORE_OR_AFTER]] VPlanTransforms::introduceMasksAndLinearize
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll b/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll
index cb463c98a61819..df41456cb841af 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_combined_exits.ll
@@ -759,3 +759,47 @@ loop:
exit:
ret void
}
+
+define void @combined_exit_conditions_one_vector_iteration_plus_one_scalar(ptr align 4 dereferenceable(80) readonly %src, ptr align 4 dereferenceable(80) noalias %dst, ptr align 4 dereferenceable(80) readonly %pred) {
+; CHECK-LABEL: define void @combined_exit_conditions_one_vector_iteration_plus_one_scalar(
+; CHECK-SAME: ptr readonly align 4 dereferenceable(80) [[SRC:%.*]], ptr noalias align 4 dereferenceable(80) [[DST:%.*]], ptr readonly align 4 dereferenceable(80) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[SRC_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[SRC]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i32, ptr [[SRC_PTR]], align 4
+; CHECK-NEXT: [[ADD:%.*]] = add nsw i32 [[DATA]], 1
+; CHECK-NEXT: [[DST_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[DST]], i64 [[IV]]
+; CHECK-NEXT: store i32 [[ADD]], ptr [[DST_PTR]], align 4
+; CHECK-NEXT: [[EE_PTR:%.*]] = getelementptr inbounds nuw [4 x i8], ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i32, ptr [[EE_PTR]], align 4
+; CHECK-NEXT: [[EE_CMP:%.*]] = icmp ne i32 [[EE_VAL]], 0
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_CMP:%.*]] = icmp eq i64 [[IV_NEXT]], 5
+; CHECK-NEXT: [[COMBINED_COND:%.*]] = select i1 [[EE_CMP]], i1 true, i1 [[COUNTED_CMP]]
+; CHECK-NEXT: br i1 [[COMBINED_COND]], label %[[EXIT:.*]], label %[[FOR_BODY]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.body ]
+ %src.ptr = getelementptr inbounds nuw [4 x i8], ptr %src, i64 %iv
+ %data = load i32, ptr %src.ptr, align 4
+ %add = add nsw i32 %data, 1
+ %dst.ptr = getelementptr inbounds nuw [4 x i8], ptr %dst, i64 %iv
+ store i32 %add, ptr %dst.ptr, align 4
+ %ee.ptr = getelementptr inbounds nuw [4 x i8], ptr %pred, i64 %iv
+ %ee.val = load i32, ptr %ee.ptr, align 4
+ %ee.cmp = icmp ne i32 %ee.val, 0
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cmp = icmp eq i64 %iv.next, 5
+ %combined.cond = select i1 %ee.cmp, i1 true, i1 %counted.cmp
+ br i1 %combined.cond, label %exit, label %for.body
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_legality.ll b/llvm/test/Transforms/LoopVectorize/early_exit_legality.ll
index 4593cd30f63a4c..f1d5e14388fe4b 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_legality.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_legality.ll
@@ -326,7 +326,7 @@ return:
; support this yet.
define i64 @uncountable_exit_on_last_block() !dbg !47 {
; CHECK-DEBUG-LABEL: LV: Checking a loop in 'uncountable_exit_on_last_block'
-; CHECK-DEBUG: LV: Not vectorizing: Cannot determine exact exit count for latch block.
+; CHECK-DEBUG: LV: Not vectorizing: Cannot determine symbolic max exit count for latch block.
; CHECK-REMARK: foo.c:100:3: loop not vectorized: Cannot vectorize early exit loop
entry:
%p1 = alloca [1024 x i8]
@@ -483,8 +483,8 @@ loop.end:
define void @exit_conditions_combined_in_single_branch(ptr noalias dereferenceable(40) %array, ptr readonly align 2 dereferenceable(40) %pred) !dbg !57 {
; CHECK-DEBUG-LABEL: LV: Checking a loop in 'exit_conditions_combined_in_single_branch'
-; CHECK-DEBUG: LV: Not vectorizing: Cannot vectorize uncountable loop.
-; CHECK-REMARK: foo.c:150:3: loop not vectorized: Cannot vectorize uncountable loop
+; CHECK-DEBUG: LV: Not vectorizing: Auto-vectorization of loops with uncountable early exit and side effects is not enabled.
+; CHECK-REMARK: foo.c:150:3: loop not vectorized: Auto-vectorization of loops with uncountable early exit and side effects is not enabled
entry:
br label %for.body, !dbg !58
@@ -511,8 +511,8 @@ exit:
; the early-exit loop is still rejected here.
define i64 @same_exit_block_with_recurrence_that_is_also_an_induction() !dbg !59 {
; CHECK-DEBUG-LABEL: LV: Checking a loop in 'same_exit_block_with_recurrence_that_is_also_an_induction'
-; CHECK-DEBUG: LV: Not vectorizing: Found reductions or recurrences in early-exit loop.
-; CHECK-REMARK: foo.c:160:3: loop not vectorized: Cannot vectorize early exit loop with reductions or recurrences
+; CHECK-DEBUG: LV: Not vectorizing: Found reductions or recurrences in uncountable exit loop.
+; CHECK-REMARK: foo.c:160:3: loop not vectorized: Cannot vectorize uncountable exit loop with reductions or recurrences
entry:
%p1 = alloca [4096 x i8]
call void @init_mem(ptr %p1, i64 4096)
@@ -542,8 +542,8 @@ loop.end:
define i64 @same_exit_block_pre_inc_use1_with_reduction() !dbg !61 {
; CHECK-DEBUG-LABEL: LV: Checking a loop in 'same_exit_block_pre_inc_use1_with_reduction'
-; CHECK-DEBUG: LV: Not vectorizing: Found reductions or recurrences in early-exit loop.
-; CHECK-REMARK: foo.c:170:3: loop not vectorized: Cannot vectorize early exit loop with reductions or recurrences
+; CHECK-DEBUG: LV: Not vectorizing: Found reductions or recurrences in uncountable exit loop.
+; CHECK-REMARK: foo.c:170:3: loop not vectorized: Cannot vectorize uncountable exit loop with reductions or recurrences
entry:
%p1 = alloca [1024 x i8]
%p2 = alloca [1024 x i8]
@@ -653,7 +653,7 @@ loop.end:
; exit count (loop is infinite without early exits).
define void @uncountable_exits_invariant_conditions(ptr %p, i1 %cond1, i1 %cond2, i1 %cond3) !dbg !67 {
; CHECK-DEBUG-LABEL: LV: Checking a loop in 'uncountable_exits_invariant_conditions'
-; CHECK-DEBUG: LV: Not vectorizing: Cannot determine exact exit count for latch block.
+; CHECK-DEBUG: LV: Not vectorizing: Cannot determine symbolic max exit count for latch block.
; CHECK-REMARK: foo.c:200:3: loop not vectorized: Cannot vectorize early exit loop
; CHECK-REMARK-NEXT: foo.c:200:3: loop not vectorized: could not determine number of loop iterations
entry:
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll b/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll
index 3a0fab8118424c..12999806432721 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll
@@ -798,7 +798,7 @@ exit:
; Vectorizeable, requires improvements in dereferenceability checks
define void @uncountable_exit_with_invariant_but_unknown_stride(ptr dereferenceable(4000) noalias %array, ptr align 2 dereferenceable(4000) readonly %pred, i64 %stride) !dbg !62 {
; CHECK-DEBUG-LABEL: LV: Checking a loop in 'uncountable_exit_with_invariant_but_unknown_stride'
-; CHECK-DEBUG: LV: Not vectorizing: Cannot determine exact exit count for latch block.
+; CHECK-DEBUG: LV: Not vectorizing: Cannot determine symbolic max exit count for latch block.
; CHECK-REMARK: foo.c:270:3: loop not vectorized: Cannot vectorize early exit loop
; CHECK-REMARK-NEXT: foo.c:270:3: loop not vectorized: could not determine number of loop iterations
entry:
diff --git a/llvm/test/Transforms/LoopVectorize/uncountable-single-exit-loops.ll b/llvm/test/Transforms/LoopVectorize/uncountable-single-exit-loops.ll
index 327c9f1668854b..71c6a34d97dfe1 100644
--- a/llvm/test/Transforms/LoopVectorize/uncountable-single-exit-loops.ll
+++ b/llvm/test/Transforms/LoopVectorize/uncountable-single-exit-loops.ll
@@ -4,14 +4,14 @@
; CHECK-LABEL: LV: Checking a loop in 'latch_exit_cannot_compute_btc_due_to_step'
; CHECK: LV: Did not find one integer induction var.
-; CHECK-NEXT: LV: Not vectorizing: Cannot determine exact exit count for latch block.
+; CHECK-NEXT: LV: Not vectorizing: Cannot determine symbolic max exit count for latch block.
; CHECK-NEXT: LV: Not vectorizing: Cannot vectorize uncountable loop.
; CHECK-NEXT: LV: Not vectorizing: Cannot prove legality.
; CHECK-LABEL: LV: Checking a loop in 'header_exit_cannot_compute_btc_due_to_step'
; CHECK: LV: Found an induction variable.
; CHECK-NEXT: LV: Did not find one integer induction var.
-; CHECK-NEXT: LV: Not vectorizing: Cannot determine exact exit count for latch block.
+; CHECK-NEXT: LV: Not vectorizing: Cannot determine symbolic max exit count for latch block.
; CHECK-NEXT: LV: Not vectorizing: Cannot vectorize uncountable loop.
; CHECK-NEXT: LV: Not vectorizing: Cannot prove legality.
More information about the llvm-commits
mailing list