[llvm] f1b42dc - [LV] Vectorize early exit loops with stores using masking (#178454)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Jun 3 07:48:38 PDT 2026
Author: Graham Hunter
Date: 2026-06-03T15:48:32+01:00
New Revision: f1b42dcc2326b80116430c9a60f41b0f86abbff7
URL: https://github.com/llvm/llvm-project/commit/f1b42dcc2326b80116430c9a60f41b0f86abbff7
DIFF: https://github.com/llvm/llvm-project/commit/f1b42dcc2326b80116430c9a60f41b0f86abbff7.diff
LOG: [LV] Vectorize early exit loops with stores using masking (#178454)
This is an alternative approach to vectorizing early exit loops with
stores that avoids needing to add an extra check block. This is a
fairly straightforward approach that should work on vector ISAs
supporting masked memory ops.
The basic approach is to create a mask covering all lanes _before_ any
exiting lane, using cttz.elts and active.lane.mask (which sets all lanes
to true if the uncountable exit wasn't taken). If the uncountable exit
was taken, then there will still be one scalar iteration left to perform
after the vector loop, which will also handle which exit block we should
branch to.
We no longer need to advance exit conditions in the vector body to the
next iteration (compared to the other PR), though we still need to move
the recipes needed to generate the exit condition (depending on which
memory operations are first in the loop).
The advantage this has over a full in-loop mask approach is that we
don't need to form intermediate masks for each uncountable exit; while I
haven't tried to mix this with the ongoing multiple-exit work yet, we
should be able to handle them without increasing the amount of generated
per-exit code. We also won't need to unpick which exit condition was met
first.
For a pseudo-C example of the transformation (with S1 and S2
representing statements with a side effect, like stores, or possibly a
load that may fault if continued past the early exit), given the
following scalar loop:
```c
for (i = 0; i < N; ++i) {
S1;
if (a[i] == threshold)
break;
S2;
}
```
we would have a vector loop and scalar tail like the following:
```c
int i = 0;
for (; i < vecN; i += VF) {
// Move load for uncountable exit condition before other
// operations in the loop.
vecA = a[i]...a[i+VF-1];
// Create mask for all lanes _before_ any uncountable exit.
vecCmp = vecA == splat(threshold);
mask = get.active.lane.mask(0, cttz.elts(vecCmp));
// Execute statements with side effects using the mask
vecS1(mask);
vecS2(mask);
// If there was an uncountable exit, increase IV by the number
// of elements in the mask, and bail out to the scalar tail.
if (any_of(vecCmp)) {
i += cttz.elts(vecCmp);
break;
}
}
// Scalar tail handles remaining iterations, plus any differences
// in exit block for different exits.
for (; i < N; ++i) {
S1;
if (a[i] == threshold)
break;
S2;
}
```
For the mask, given a comparison result of `<0, 0, 1, 0>`, we would
expect a mask of `<1, 1, 0, 0>`.
Added:
llvm/test/Transforms/LoopVectorize/interleave_uncountable_exits.ll
llvm/test/Transforms/LoopVectorize/scalarized_conditional_ops_uncountable_exits.ll
llvm/test/Transforms/LoopVectorize/tail_fold_uncountable_exits.ll
Modified:
llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
llvm/lib/Transforms/Vectorize/VPlanTransforms.h
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
llvm/test/Transforms/LoopVectorize/AArch64/early_exit_with_stores.ll
llvm/test/Transforms/LoopVectorize/RISCV/early_exit_with_stores.ll
llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll
llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll
llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll
llvm/unittests/Transforms/Vectorize/VPlanUncountableExitTest.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index 897faafd8957e..214b5cff8a03a 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -1836,7 +1836,11 @@ bool LoopVectorizationLegality::canUncountableExitConditionLoadBeMoved(
if (&I == Load)
continue;
- if (I.mayWriteToMemory()) {
+ if (I.mayReadOrWriteMemory()) {
+ // We need to mask all other memory ops.
+ ConditionallyExecutedOps.insert(&I);
+ if (isa<LoadInst>(&I))
+ continue;
if (auto *SI = dyn_cast<StoreInst>(&I)) {
AliasResult AR = AA->alias(Ptr, SI->getPointerOperand());
if (AR == AliasResult::NoAlias)
@@ -1942,12 +1946,23 @@ bool LoopVectorizationLegality::canVectorize(bool UseVPlanNativePath) {
return false;
}
- // Bail out for ReadWrite loops with uncountable exits for now.
- if (UncountableExitType == UncountableExitTrait::ReadWrite) {
- reportVectorizationFailure(
- "Writes to memory unsupported in early exit loops",
- "Cannot vectorize early exit loop with writes to memory",
- "WritesInEarlyExitLoop", ORE, TheLoop);
+ // TODO: Remove this restriction once we're sure it's safe to do so.
+ // Handling stores to invariant addresses will be slightly
diff erent
+ // based on the vectorization style chosen. If we bail out to a scalar
+ // tail before executing any lane that would take the uncountable exit,
+ // then the store that occurs in the scalar loop would suffice.
+ //
+ // If we instead handle the lane taking the uncountable exit within the
+ // vectorized loop, then we will have to ensure that we extract the
+ // last active lane at that point in the loop instead of the last lane
+ // of the vector before performing a scalar store.
+ if (UncountableExitType != UncountableExitTrait::None &&
+ !LAI->getStoresToInvariantAddresses().empty()) {
+ LLVM_DEBUG(dbgs() << "LV: Cannot vectorize early exit loops with stores to "
+ "loop-invariant addresses\n");
+ reportVectorizationFailure("Cannot vectorize early exit loops with stores "
+ "to loop-invariant addresses",
+ "LoopInvariantStoresInEELoop", ORE, TheLoop);
return false;
}
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index a5fb60a3232e7..fcfe11b6e9de6 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -403,6 +403,12 @@ static cl::opt<bool> EnableEarlyExitVectorization(
cl::desc(
"Enable vectorization of early exit loops with uncountable exits."));
+static cl::opt<bool> EnableEarlyExitVectorizationWithSideEffects(
+ "enable-early-exit-vectorization-with-side-effects", cl::init(false),
+ cl::Hidden,
+ cl::desc("Enable vectorization of early exit loops with uncountable exits "
+ "and side effects"));
+
// Likelyhood of bypassing the vectorized loop because there are zero trips left
// after prolog. See `emitIterationCountCheck`.
static constexpr uint32_t MinItersBypassWeights[] = {1, 127};
@@ -2449,8 +2455,9 @@ bool LoopVectorizationCostModel::isPredicatedInst(Instruction *I) const {
if (Legal->blockNeedsPredication(I->getParent()))
return true;
- // If we're not folding the tail by masking, predication is unnecessary.
- if (!foldTailByMasking())
+ // If we're not folding the tail by masking and not vectorizing a loop with
+ // uncountable exits and side effects, predication is unnecessary.
+ if (!foldTailByMasking() && !Legal->hasUncountableExitWithSideEffects())
return false;
// All that remain are instructions with side-effects originally executed in
@@ -6546,8 +6553,12 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
return nullptr;
}
- RUN_VPLAN_PASS(VPlanTransforms::addMiddleCheck, *VPlan0,
- CM.foldTailByMasking());
+ // If we're handling uncountable exits in the scalar tail after a vector
+ // loop with an in-loop mask, then the middle check has already been
+ // created to compare against the actual number of lanes executed.
+ if (EEStyle != UncountableExitStyle::MaskedHandleExitInScalarLoop)
+ RUN_VPLAN_PASS(VPlanTransforms::addMiddleCheck, *VPlan0,
+ CM.foldTailByMasking());
RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
if (CM.foldTailByMasking())
@@ -7878,6 +7889,14 @@ bool LoopVectorizePass::processLoop(Loop *L) {
"UncountableEarlyExitLoopsDisabled", ORE, L);
return false;
}
+ if (LVL.hasUncountableExitWithSideEffects() &&
+ !EnableEarlyExitVectorizationWithSideEffects) {
+ reportVectorizationFailure("Auto-vectorization of loops with uncountable "
+ "early exit and side effects is not enabled",
+ "UncountableEarlyExitSideEffectLoopsDisabled",
+ ORE, L);
+ return false;
+ }
}
InterleavedAccessInfo IAI(PSE, L, DT, LI, LVL.getLAI());
@@ -8144,6 +8163,15 @@ bool LoopVectorizePass::processLoop(Loop *L) {
IC = 1;
}
+ // FIXME: Enable interleaving for EE-with-side-effects.
+ if (InterleaveLoop && LVL.hasUncountableExitWithSideEffects()) {
+ LLVM_DEBUG(dbgs() << "LV: Not interleaving due to EE with side effects.\n");
+ IntDiagMsg = {"EEWithSideEffectsPreventsInterleaving",
+ "Unable to interleave due to early exit with side effects."};
+ InterleaveLoop = false;
+ IC = 1;
+ }
+
// Emit diagnostic messages, if any.
if (!VectorizeLoop && !InterleaveLoop) {
// Do not vectorize or interleaving the loop.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index e5c546c7fb851..d6b70068fe7f9 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -1265,11 +1265,15 @@ bool VPlanTransforms::handleEarlyExits(VPlan &Plan, UncountableExitStyle Style,
// here from handleUncountableEarlyExits, but we need to improve
// detection of recipes which may write to memory.
if (Style != UncountableExitStyle::NoUncountableExit) {
- if (!areAllLoadsDereferenceable(HeaderVPBB, TheLoop, PSE, DT, AC))
+ // Dereferenceability is checked separately for uncountable exit loops with
+ // stores, as only the loads contributing to the exit condition need to
+ // be checked.
+ if (Style == UncountableExitStyle::ReadOnly &&
+ !areAllLoadsDereferenceable(HeaderVPBB, TheLoop, PSE, DT, AC))
return false;
// TODO: Check target preference for style.
- handleUncountableEarlyExits(Plan, HeaderVPBB, LatchVPBB, MiddleVPBB, Style);
- return true;
+ return handleUncountableEarlyExits(Plan, HeaderVPBB, LatchVPBB, MiddleVPBB,
+ TheLoop, PSE, DT, AC, Style);
}
// Disconnect countable early exits from the loop, leaving it with a single
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 24d411d6e1914..62d56fa762874 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -30,6 +30,7 @@
#include "llvm/ADT/TypeSwitch.h"
#include "llvm/Analysis/IVDescriptors.h"
#include "llvm/Analysis/InstSimplifyFolder.h"
+#include "llvm/Analysis/Loads.h"
#include "llvm/Analysis/LoopInfo.h"
#include "llvm/Analysis/MemoryLocation.h"
#include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
@@ -3987,6 +3988,19 @@ void VPlanTransforms::expandBranchOnTwoConds(VPlan &Plan) {
VPBlockBase *Succ0 = Successors[0];
VPBlockBase *Succ1 = Successors[1];
VPBlockBase *Succ2 = Successors[2];
+
+ // If the successor block for both conditions is the same, then combine the
+ // two conditions and plant a single conditional branch.
+ if (Succ0 == Succ1) {
+ VPBuilder Builder(Br);
+ VPValue *Combined = Builder.createOr(Cond0, Cond1, DL);
+ Builder.createNaryOp(VPInstruction::BranchOnCond, {Combined}, DL);
+ VPBlockUtils::connectBlocks(BrOnTwoCondsBB, Succ0);
+ VPBlockUtils::connectBlocks(BrOnTwoCondsBB, Succ2);
+ Br->eraseFromParent();
+ continue;
+ }
+
assert(!Succ0->getParent() && !Succ1->getParent() && !Succ2->getParent() &&
!BrOnTwoCondsBB->getParent() && "regions must already be dissolved");
@@ -4188,19 +4202,174 @@ void VPlanTransforms::convertToConcreteRecipes(VPlan &Plan) {
}
}
-void VPlanTransforms::handleUncountableEarlyExits(VPlan &Plan,
- VPBasicBlock *HeaderVPBB,
- VPBasicBlock *LatchVPBB,
- VPBasicBlock *MiddleVPBB,
- UncountableExitStyle Style) {
- struct EarlyExitInfo {
- VPBasicBlock *EarlyExitingVPBB;
- VPIRBasicBlock *EarlyExitVPBB;
- VPValue *CondToExit;
- };
+struct EarlyExitInfo {
+ VPBasicBlock *EarlyExitingVPBB;
+ VPIRBasicBlock *EarlyExitVPBB;
+ VPValue *CondToExit;
+};
+
+/// Update \p Plan to mask memory operations in the loop based on whether the
+/// early exit is taken or not.
+static bool handleUncountableExitsWithSideEffects(
+ VPlan &Plan, SmallVectorImpl<EarlyExitInfo> &Exits,
+ VPBasicBlock *HeaderVPBB, VPBasicBlock *LatchVPBB, VPBasicBlock *MiddleVPBB,
+ Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT,
+ AssumptionCache *AC, VPDominatorTree &VPDT) {
+
+ // Disconnect early exiting blocks from successors, remove branches. We
+ // currently don't support multiple uses for recipes involved in creating
+ // the uncountable exit condition.
+ for (auto &Exit : Exits) {
+ if (Exit.EarlyExitingVPBB == LatchVPBB)
+ continue;
+
+ for (VPRecipeBase &R : Exit.EarlyExitVPBB->phis())
+ cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
+ Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
+ VPBlockUtils::disconnectBlocks(Exit.EarlyExitingVPBB, Exit.EarlyExitVPBB);
+ }
+
+ // We can abandon a VPlan entirely if we return false here, so we shouldn't
+ // crash if some earlier assumptions on scalar IR don't hold for the vplan
+ // version of the loop.
+ SmallVector<VPInstruction *, 2> GEPs;
+ SmallVector<VPInstruction *, 8> ConditionRecipes;
+
+ std::optional<VPValue *> Cond =
+ vputils::getRecipesForUncountableExit(ConditionRecipes, GEPs, LatchVPBB);
+ if (!Cond)
+ return false;
+
+ // Find load contributing to condition.
+ VPRecipeBase *CondLoad = nullptr;
+ for (auto *Recipe : ConditionRecipes) {
+ if (match(Recipe, m_VPInstruction<Instruction::Load>(m_VPValue()))) {
+ // TODO: Support more than one load. Needs legality updates too.
+ assert(CondLoad == nullptr && "Too many condition loads");
+ CondLoad = Recipe;
+ }
+ }
+ assert(CondLoad && "Couldn't find load");
+
+ // Ensure that we are guaranteed to be able to dereference the memory used
+ // for determining the uncountable exit for the maximum possible number of
+ // scalar iterations of the loop.
+ //
+ // TODO: Support first-faulting loads in cases where we don't know whether
+ // all possible addresses are dereferenceable.
+ {
+ SmallVector<const SCEVPredicate *, 4> Predicates;
+ VPSingleDefRecipe *Load = cast<VPSingleDefRecipe>(CondLoad);
+ VPValue *Ptr = Load->getOperand(0);
+ const SCEV *PtrSCEV = vputils::getSCEVExprForVPValue(Ptr, PSE, TheLoop);
+ const DataLayout &DL = Plan.getDataLayout();
+ APInt EltSize(DL.getIndexTypeSizeInBits(Ptr->getScalarType()),
+ DL.getTypeStoreSize(Load->getScalarType()).getFixedValue());
+ if (!isDereferenceableAndAlignedInLoop(
+ PtrSCEV, cast<LoadInst>(Load->getUnderlyingInstr())->getAlign(),
+ PSE.getSE()->getConstant(EltSize), TheLoop, *PSE.getSE(), DT, AC,
+ &Predicates))
+ return false;
+ }
+
+ // Check GEPs to see if we can link them to a widen IV recipe with a step of
+ // 1; we're only interested in contiguous accesses for the condition load
+ // right now.
+ for (auto *GEP : GEPs) {
+ VPValue *MaybeIV = nullptr;
+ if (!match(GEP, m_VPInstruction<Instruction::GetElementPtr>(
+ m_LiveIn(), m_VPValue(MaybeIV))))
+ return false;
+
+ auto *WIV = dyn_cast<VPWidenInductionRecipe>(MaybeIV);
+ if (!WIV)
+ return false;
+
+ if (!match(WIV->getStartValue(), m_SpecificInt(0)) ||
+ !match(WIV->getStepValue(), m_SpecificInt(1)))
+ return false;
+ }
+
+ // Find an insertion point. Default to the end of the header but override
+ // if we find a memory op that needs masking before the condition load.
+ auto InsertIt = HeaderVPBB->end();
+ VPRecipeBase *CondR = (*Cond)->getDefiningRecipe();
+ bool CondMoveNeeded = CondR->getParent() != HeaderVPBB;
+ for (VPRecipeBase &R : *HeaderVPBB) {
+ if (&R == CondLoad)
+ continue;
+
+ if (R.mayReadOrWriteMemory()) {
+ if (!VPDT.properlyDominates(CondR, &R)) {
+ CondMoveNeeded = true;
+ InsertIt = R.getIterator();
+ }
+ break;
+ }
+ }
+
+ // If another memory operation would take place before the comparison to
+ // determine whether to exit early or the comparison doesn't take place in
+ // the header, move the comparison (and supporting recipes).
+ if (CondMoveNeeded)
+ for (auto *Recipe : reverse(ConditionRecipes))
+ Recipe->moveBefore(*HeaderVPBB, InsertIt);
+
+ // Create a mask to represent all lanes that fully execute in the vector loop,
+ // stopping short of any early exit.
+ VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
+ VPValue *FirstActive = MaskBuilder.createFirstActiveLane(*Cond);
+ VPValue *IV = cast<VPSingleDefRecipe>(&HeaderVPBB->front());
+ Type *IVScalarTy = IV->getScalarType();
+ Type *FirstActiveTy = FirstActive->getScalarType();
+ VPValue *ALMMultiplier = Plan.getConstantInt(IVScalarTy, 1);
+ VPValue *Zero = Plan.getZero(IVScalarTy);
+ FirstActive = MaskBuilder.createScalarZExtOrTrunc(FirstActive, IVScalarTy,
+ FirstActiveTy, DebugLoc());
+ VPValue *Mask = MaskBuilder.createNaryOp(VPInstruction::ActiveLaneMask,
+ {Zero, FirstActive, ALMMultiplier},
+ DebugLoc(), "uncountable.exit.mask");
+
+ // Convert all other memory operations to use the mask.
+ for (VPBasicBlock *VPBB : vp_rpo_plain_cfg_loop_body(HeaderVPBB))
+ for (VPRecipeBase &R : *VPBB)
+ if (R.mayReadOrWriteMemory() && &R != CondLoad) {
+ // TODO: Handle conditional memory operations in the loop.
+ if (!VPDT.dominates(R.getParent(), LatchVPBB))
+ return false;
+ cast<VPInstruction>(&R)->addMask(Mask);
+ }
+
+ // Update middle block branch to compare (IV + however many lanes were active)
+ // against the full trip count, since we may be exiting the vector loop early.
+ // If we didn't take an early exit, we should get the equivalent of VF from
+ // the FirstActiveLane.
+ VPBuilder MiddleBuilder(MiddleVPBB, MiddleVPBB->end());
+ VPValue *ScalarIV = MiddleBuilder.createNaryOp(VPInstruction::ExtractLane,
+ {Zero, IV}, DebugLoc());
+ VPValue *ExitIV = MiddleBuilder.createAdd(ScalarIV, FirstActive);
+ VPValue *FullTC =
+ MiddleBuilder.createICmp(CmpInst::ICMP_EQ, ExitIV, Plan.getTripCount());
+ MiddleBuilder.createNaryOp(VPInstruction::BranchOnCond, {FullTC});
+
+ // Update resume phi in scalar.ph.
+ VPBasicBlock *ScalarPH = Plan.getScalarPreheader();
+ auto Phis = ScalarPH->phis();
+ // TODO: Handle more than one Phi; re-derive from IV.
+ // TODO: Handle reductions.
+ if (range_size(Phis) != 1)
+ return false;
+ VPPhi *ContinueIV = cast<VPPhi>(Phis.begin());
+ ContinueIV->setOperand(0, ExitIV);
+ return true;
+}
+bool VPlanTransforms::handleUncountableEarlyExits(
+ VPlan &Plan, VPBasicBlock *HeaderVPBB, VPBasicBlock *LatchVPBB,
+ VPBasicBlock *MiddleVPBB, Loop *TheLoop, PredicatedScalarEvolution &PSE,
+ DominatorTree &DT, AssumptionCache *AC, UncountableExitStyle Style) {
VPDominatorTree VPDT(Plan);
- VPBuilder Builder(LatchVPBB->getTerminator());
+ VPBuilder LatchBuilder(LatchVPBB->getTerminator());
SmallVector<EarlyExitInfo> Exits;
for (VPIRBasicBlock *ExitBlock : Plan.getExitBlocks()) {
for (VPBlockBase *Pred : to_vector(ExitBlock->getPredecessors())) {
@@ -4265,13 +4434,36 @@ void VPlanTransforms::handleUncountableEarlyExits(VPlan &Plan,
// exit is taken.
VPValue *Combined = Exits[0].CondToExit;
for (const EarlyExitInfo &Info : drop_begin(Exits))
- Combined = Builder.createLogicalOr(Combined, Info.CondToExit);
+ Combined = LatchBuilder.createLogicalOr(Combined, Info.CondToExit);
VPValue *IsAnyExitTaken =
- Builder.createNaryOp(VPInstruction::AnyOf, {Combined});
+ LatchBuilder.createNaryOp(VPInstruction::AnyOf, {Combined});
- assert(Style == UncountableExitStyle::ReadOnly &&
- "Early exit store masking not implemented");
+ // Create a comparison for the latch exit condition and replace the
+ // BranchOnCond with a BranchOnTwoConds. The original BranchOnCond's condition
+ // is used as the latch-exit condition; canonical IV recipes have not been
+ // introduced yet, so there is no BranchOnCount to derive the condition from.
+ auto *LatchExitingBranch = cast<VPInstruction>(LatchVPBB->getTerminator());
+ assert(LatchExitingBranch->getOpcode() == VPInstruction::BranchOnCond &&
+ "Unexpected terminator");
+ VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
+ DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
+ LatchExitingBranch->eraseFromParent();
+ LatchBuilder.setInsertPoint(LatchVPBB);
+ LatchBuilder.createNaryOp(VPInstruction::BranchOnTwoConds,
+ {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
+ LatchVPBB->clearSuccessors();
+
+ if (Style == UncountableExitStyle::MaskedHandleExitInScalarLoop) {
+ // If handling the exiting lane in the scalar loop, combine the exit
+ // conditions into a single BranchOnCond.
+ LatchVPBB->setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
+ MiddleVPBB->clearPredecessors();
+ MiddleVPBB->setPredecessors({LatchVPBB, LatchVPBB});
+ return handleUncountableExitsWithSideEffects(Plan, Exits, HeaderVPBB,
+ LatchVPBB, MiddleVPBB, TheLoop,
+ PSE, DT, AC, VPDT);
+ }
// Create the vector.early.exit blocks.
SmallVector<VPBasicBlock *> VectorEarlyExitVPBBs(Exits.size());
@@ -4290,6 +4482,7 @@ void VPlanTransforms::handleUncountableEarlyExits(VPlan &Plan,
Exits.size() == 1 ? VectorEarlyExitVPBBs[0]
: Plan.createVPBasicBlock("vector.early.exit.check");
DispatchVPBB->setPredecessors({LatchVPBB});
+ LatchVPBB->setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
VPBuilder DispatchBuilder(DispatchVPBB, DispatchVPBB->begin());
VPValue *FirstActiveLane = DispatchBuilder.createFirstActiveLane(
{Combined}, DebugLoc::getUnknown(), "first.active.lane");
@@ -4396,22 +4589,7 @@ void VPlanTransforms::handleUncountableEarlyExits(VPlan &Plan,
DispatchBuilder.setInsertPoint(CurrentBB);
}
- // Replace the latch terminator with the new branching logic. The original
- // BranchOnCond's condition is used as the latch-exit condition; canonical IV
- // recipes have not been introduced yet, so there is no BranchOnCount to
- // derive the condition from.
- auto *LatchExitingBranch = cast<VPInstruction>(LatchVPBB->getTerminator());
- assert(LatchExitingBranch->getOpcode() == VPInstruction::BranchOnCond &&
- "Unexpected terminator");
- VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
-
- DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
- LatchExitingBranch->eraseFromParent();
- Builder.setInsertPoint(LatchVPBB);
- Builder.createNaryOp(VPInstruction::BranchOnTwoConds,
- {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
- LatchVPBB->clearSuccessors();
- LatchVPBB->setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
+ return true;
}
/// This function tries convert extended in-loop reductions to
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index b4676cef6a199..961931798b9c8 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -346,10 +346,10 @@ struct VPlanTransforms {
/// appropriate branching logic in the latch that handles early exits and the
/// latch exit condition. Multiple exits are handled with a dispatch block
/// that determines which exit to take based on lane-by-lane semantics.
- static void handleUncountableEarlyExits(VPlan &Plan, VPBasicBlock *HeaderVPBB,
- VPBasicBlock *LatchVPBB,
- VPBasicBlock *MiddleVPBB,
- UncountableExitStyle Style);
+ static bool handleUncountableEarlyExits(
+ VPlan &Plan, VPBasicBlock *HeaderVPBB, VPBasicBlock *LatchVPBB,
+ VPBasicBlock *MiddleVPBB, Loop *TheLoop, PredicatedScalarEvolution &PSE,
+ DominatorTree &DT, AssumptionCache *AC, UncountableExitStyle Style);
/// Replaces the exit condition from
/// (branch-on-cond eq CanonicalIVInc, VectorTripCount)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 17a825429d89b..f7f2d4591a3ce 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -569,9 +569,8 @@ vputils::getRecipesForUncountableExit(SmallVectorImpl<VPInstruction *> &Recipes,
// EMIT vp<%index.next> = add nuw vp<%2>, vp<%0>
// EMIT vp<%4> = any-of ir<%3>
// EMIT vp<%5> = icmp eq vp<%index.next>, vp<%1>
- // EMIT vp<%6> = or vp<%4>, vp<%5>
- // EMIT branch-on-cond vp<%6>
- // Successor(s): middle.block, for.body
+ // EMIT branch-on-two-conds vp<%4>, vp<%5>
+ // Successor(s): middle.block, middle.block, for.body
//
// middle.block:
// Successor(s): ir-bb<exit>, scalar.ph
@@ -585,8 +584,8 @@ vputils::getRecipesForUncountableExit(SmallVectorImpl<VPInstruction *> &Recipes,
// Find the uncountable loop exit condition.
VPValue *UncountableCondition = nullptr;
if (!match(LatchVPBB->getTerminator(),
- m_BranchOnCond(m_c_BinaryOr(
- m_AnyOf(m_VPValue(UncountableCondition)), m_VPValue()))))
+ m_BranchOnTwoConds(m_AnyOf(m_VPValue(UncountableCondition)),
+ m_VPValue())))
return std::nullopt;
SmallVector<VPValue *, 4> Worklist;
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/early_exit_with_stores.ll b/llvm/test/Transforms/LoopVectorize/AArch64/early_exit_with_stores.ll
index e2fd747cc2dbf..46a21f8a95079 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/early_exit_with_stores.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/early_exit_with_stores.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; RUN: opt -S < %s -p loop-vectorize -mattr=+sve | FileCheck %s
+; RUN: opt -S < %s -p loop-vectorize -mattr=+sve -enable-early-exit-vectorization-with-side-effects | FileCheck %s
target triple = "aarch64-unknown-linux-gnu"
@@ -52,10 +52,42 @@ loop.end:
define void @loop_contains_store_condition_load_has_single_user(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: define void @loop_contains_store_condition_load_has_single_user(
; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 20, [[TMP1]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 20, [[TMP3]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 20, [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[FOR_BODY1]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 8 x i16>, ptr [[TMP5]], align 2
+; CHECK-NEXT: [[TMP6:%.*]] = icmp sgt <vscale x 8 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP6]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 0, i64 [[TMP7]])
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i16> @llvm.masked.load.nxv8i16.p0(ptr align 2 [[ST_ADDR1]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]], <vscale x 8 x i16> poison)
+; CHECK-NEXT: [[TMP8:%.*]] = add nsw <vscale x 8 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.nxv8i16.p0(<vscale x 8 x i16> [[TMP8]], ptr align 2 [[ST_ADDR1]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP9:%.*]] = freeze <vscale x 8 x i1> [[TMP6]]
+; CHECK-NEXT: [[TMP10:%.*]] = call i1 @llvm.vector.reduce.or.nxv8i1(<vscale x 8 x i1> [[TMP9]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV1]], [[TMP3]]
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: [[TMP12:%.*]] = or i1 [[TMP10]], [[TMP11]]
+; CHECK-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[FOR_BODY1]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP13:%.*]] = add i64 [[IV1]], [[TMP7]]
+; CHECK-NEXT: [[TMP14:%.*]] = icmp eq i64 [[TMP13]], 20
+; CHECK-NEXT: br i1 [[TMP14]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[TMP13]], %[[MIDDLE_BLOCK]] ], [ 0, %[[SCALAR_PH1]] ]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
@@ -63,11 +95,11 @@ define void @loop_contains_store_condition_load_has_single_user(ptr dereferencea
; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
-; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -253,21 +285,21 @@ define void @loop_contains_store_assumed_bounds(ptr noalias %array, ptr readonly
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[N_BYTES:%.*]] = mul nuw nsw i64 [[N]], 2
; CHECK-NEXT: call void @llvm.assume(i1 true) [ "align"(ptr [[PRED]], i64 2), "dereferenceable"(ptr [[PRED]], i64 [[N_BYTES]]) ]
-; CHECK-NEXT: br label %[[FOR_BODY:.*]]
-; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR1]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -300,22 +332,55 @@ define void @loop_contains_store_to_pointer_with_no_deref_info(ptr align 2 deref
; CHECK-LABEL: define void @loop_contains_store_to_pointer_with_no_deref_info(
; CHECK-SAME: ptr readonly align 2 dereferenceable(40) [[LOAD_ARRAY:%.*]], ptr noalias align 2 [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 20, [[TMP1]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH1:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 20, [[TMP3]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 20, [[N_MOD_VF]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[LD_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[LOAD_ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[LD_ADDR]], align 2
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[LD_ADDR:%.*]] = getelementptr i16, ptr [[LOAD_ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 8 x i16>, ptr [[TMP5]], align 2
+; CHECK-NEXT: [[TMP6:%.*]] = icmp sgt <vscale x 8 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP6]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 0, i64 [[TMP7]])
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i16> @llvm.masked.load.nxv8i16.p0(ptr align 2 [[LD_ADDR]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]], <vscale x 8 x i16> poison)
+; CHECK-NEXT: [[TMP8:%.*]] = add nsw <vscale x 8 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: call void @llvm.masked.store.nxv8i16.p0(<vscale x 8 x i16> [[TMP8]], ptr align 2 [[TMP9]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP10:%.*]] = freeze <vscale x 8 x i1> [[TMP6]]
+; CHECK-NEXT: [[TMP11:%.*]] = call i1 @llvm.vector.reduce.or.nxv8i1(<vscale x 8 x i1> [[TMP10]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], [[TMP3]]
+; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: [[TMP13:%.*]] = or i1 [[TMP11]], [[TMP12]]
+; CHECK-NEXT: br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[FOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP14:%.*]] = add i64 [[IV]], [[TMP7]]
+; CHECK-NEXT: [[TMP15:%.*]] = icmp eq i64 [[TMP14]], 20
+; CHECK-NEXT: br i1 [[TMP15]], label %[[EXIT:.*]], label %[[SCALAR_PH1]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[TMP14]], %[[MIDDLE_BLOCK]] ], [ 0, %[[SCALAR_PH]] ]
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[LD_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[LOAD_ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[LD_ADDR1]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
-; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -346,22 +411,22 @@ exit:
define void @loop_contains_store_unknown_bounds(ptr align 2 dereferenceable(100) noalias %array, ptr align 2 dereferenceable(100) readonly %pred, i64 %n) {
; CHECK-LABEL: define void @loop_contains_store_unknown_bounds(
; CHECK-SAME: ptr noalias align 2 dereferenceable(100) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(100) [[PRED:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ENTRY:.*]]:
-; CHECK-NEXT: br label %[[FOR_BODY:.*]]
-; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR1]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -391,14 +456,46 @@ exit:
define void @loop_contains_store_in_latch_block(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: define void @loop_contains_store_in_latch_block(
; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[SCALAR_PH1:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 20, [[TMP1]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 20, [[TMP3]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 20, [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[FOR_BODY1]] ]
+; CHECK-NEXT: [[EE_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 8 x i16>, ptr [[EE_ADDR1]], align 2
+; CHECK-NEXT: [[TMP5:%.*]] = icmp sgt <vscale x 8 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP6:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP5]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 0, i64 [[TMP6]])
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i16> @llvm.masked.load.nxv8i16.p0(ptr align 2 [[TMP7]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]], <vscale x 8 x i16> poison)
+; CHECK-NEXT: [[TMP8:%.*]] = add nsw <vscale x 8 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.nxv8i16.p0(<vscale x 8 x i16> [[TMP8]], ptr align 2 [[TMP7]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP9:%.*]] = freeze <vscale x 8 x i1> [[TMP5]]
+; CHECK-NEXT: [[TMP10:%.*]] = call i1 @llvm.vector.reduce.or.nxv8i1(<vscale x 8 x i1> [[TMP9]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV1]], [[TMP3]]
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: [[TMP12:%.*]] = or i1 [[TMP10]], [[TMP11]]
+; CHECK-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[FOR_BODY1]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP13:%.*]] = add i64 [[IV1]], [[TMP6]]
+; CHECK-NEXT: [[TMP14:%.*]] = icmp eq i64 [[TMP13]], 20
+; CHECK-NEXT: br i1 [[TMP14]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[TMP13]], %[[MIDDLE_BLOCK]] ], [ 0, %[[SCALAR_PH1]] ]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
-; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
@@ -406,7 +503,7 @@ define void @loop_contains_store_in_latch_block(ptr dereferenceable(40) noalias
; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -1032,10 +1129,42 @@ exit:
define i32 @uncountable_exit_with_separate_exit_block(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: define i32 @uncountable_exit_with_separate_exit_block(
; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 20, [[TMP1]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH1:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 20, [[TMP3]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 20, [[N_MOD_VF]]
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[FOR_BODY1]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 8 x i16>, ptr [[TMP5]], align 2
+; CHECK-NEXT: [[TMP6:%.*]] = icmp sgt <vscale x 8 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP6]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 0, i64 [[TMP7]])
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i16> @llvm.masked.load.nxv8i16.p0(ptr align 2 [[ST_ADDR1]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]], <vscale x 8 x i16> poison)
+; CHECK-NEXT: [[TMP8:%.*]] = add nsw <vscale x 8 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.nxv8i16.p0(<vscale x 8 x i16> [[TMP8]], ptr align 2 [[ST_ADDR1]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP9:%.*]] = freeze <vscale x 8 x i1> [[TMP6]]
+; CHECK-NEXT: [[TMP10:%.*]] = call i1 @llvm.vector.reduce.or.nxv8i1(<vscale x 8 x i1> [[TMP9]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV1]], [[TMP3]]
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: [[TMP12:%.*]] = or i1 [[TMP10]], [[TMP11]]
+; CHECK-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[FOR_BODY1]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP13:%.*]] = add i64 [[IV1]], [[TMP7]]
+; CHECK-NEXT: [[TMP14:%.*]] = icmp eq i64 [[TMP13]], 20
+; CHECK-NEXT: br i1 [[TMP14]], label %[[EXIT_COUNTABLE:.*]], label %[[SCALAR_PH1]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[TMP13]], %[[MIDDLE_BLOCK]] ], [ 0, %[[SCALAR_PH]] ]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
@@ -1047,7 +1176,7 @@ define i32 @uncountable_exit_with_separate_exit_block(ptr dereferenceable(40) no
; CHECK: [[FOR_INC]]:
; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT_COUNTABLE:.*]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT_COUNTABLE]], label %[[FOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
; CHECK: [[EXIT_COUNTABLE]]:
; CHECK-NEXT: ret i32 0
; CHECK: [[EXIT_UNCOUNTABLE]]:
diff --git a/llvm/test/Transforms/LoopVectorize/RISCV/early_exit_with_stores.ll b/llvm/test/Transforms/LoopVectorize/RISCV/early_exit_with_stores.ll
index e1361126b6331..2b73a995d73bb 100644
--- a/llvm/test/Transforms/LoopVectorize/RISCV/early_exit_with_stores.ll
+++ b/llvm/test/Transforms/LoopVectorize/RISCV/early_exit_with_stores.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; RUN: opt -S < %s -p loop-vectorize -mtriple=riscv64 -mattr=+v | FileCheck %s
+; RUN: opt -S < %s -p loop-vectorize -mtriple=riscv64 -mattr=+v -enable-early-exit-vectorization-with-side-effects | FileCheck %s
;; See ../early_exit_store_legality.ll for reasons why a particular loop doesn't
;; vectorize yet.
@@ -51,21 +51,53 @@ define void @loop_contains_store_condition_load_has_single_user(ptr dereferencea
; CHECK-LABEL: define void @loop_contains_store_condition_load_has_single_user(
; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 20, [[TMP1]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH1:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 20, [[TMP3]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 20, [[N_MOD_VF]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 8 x i16>, ptr [[TMP5]], align 2
+; CHECK-NEXT: [[TMP6:%.*]] = icmp sgt <vscale x 8 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP6]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 0, i64 [[TMP7]])
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i16> @llvm.masked.load.nxv8i16.p0(ptr align 2 [[ST_ADDR]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]], <vscale x 8 x i16> poison)
+; CHECK-NEXT: [[TMP8:%.*]] = add nsw <vscale x 8 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.nxv8i16.p0(<vscale x 8 x i16> [[TMP8]], ptr align 2 [[ST_ADDR]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP9:%.*]] = freeze <vscale x 8 x i1> [[TMP6]]
+; CHECK-NEXT: [[TMP10:%.*]] = call i1 @llvm.vector.reduce.or.nxv8i1(<vscale x 8 x i1> [[TMP9]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], [[TMP3]]
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: [[TMP12:%.*]] = or i1 [[TMP10]], [[TMP11]]
+; CHECK-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[FOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP13:%.*]] = add i64 [[IV]], [[TMP7]]
+; CHECK-NEXT: [[TMP14:%.*]] = icmp eq i64 [[TMP13]], 20
+; CHECK-NEXT: br i1 [[TMP14]], label %[[EXIT:.*]], label %[[SCALAR_PH1]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[TMP13]], %[[MIDDLE_BLOCK]] ], [ 0, %[[SCALAR_PH]] ]
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR1]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
-; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -251,21 +283,21 @@ define void @loop_contains_store_assumed_bounds(ptr noalias %array, ptr readonly
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[N_BYTES:%.*]] = mul nuw nsw i64 [[N]], 2
; CHECK-NEXT: call void @llvm.assume(i1 true) [ "align"(ptr [[PRED]], i64 2), "dereferenceable"(ptr [[PRED]], i64 [[N_BYTES]]) ]
-; CHECK-NEXT: br label %[[FOR_BODY:.*]]
-; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR1]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -298,22 +330,55 @@ define void @loop_contains_store_to_pointer_with_no_deref_info(ptr align 2 deref
; CHECK-LABEL: define void @loop_contains_store_to_pointer_with_no_deref_info(
; CHECK-SAME: ptr readonly align 2 dereferenceable(40) [[LOAD_ARRAY:%.*]], ptr noalias align 2 [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 20, [[TMP1]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH1:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 20, [[TMP3]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 20, [[N_MOD_VF]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[LD_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[LOAD_ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[LD_ADDR]], align 2
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[LD_ADDR:%.*]] = getelementptr i16, ptr [[LOAD_ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 8 x i16>, ptr [[TMP5]], align 2
+; CHECK-NEXT: [[TMP6:%.*]] = icmp sgt <vscale x 8 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP6]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 0, i64 [[TMP7]])
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i16> @llvm.masked.load.nxv8i16.p0(ptr align 2 [[LD_ADDR]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]], <vscale x 8 x i16> poison)
+; CHECK-NEXT: [[TMP8:%.*]] = add nsw <vscale x 8 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: [[TMP9:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: call void @llvm.masked.store.nxv8i16.p0(<vscale x 8 x i16> [[TMP8]], ptr align 2 [[TMP9]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP10:%.*]] = freeze <vscale x 8 x i1> [[TMP6]]
+; CHECK-NEXT: [[TMP11:%.*]] = call i1 @llvm.vector.reduce.or.nxv8i1(<vscale x 8 x i1> [[TMP10]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], [[TMP3]]
+; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: [[TMP13:%.*]] = or i1 [[TMP11]], [[TMP12]]
+; CHECK-NEXT: br i1 [[TMP13]], label %[[MIDDLE_BLOCK:.*]], label %[[FOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP14:%.*]] = add i64 [[IV]], [[TMP7]]
+; CHECK-NEXT: [[TMP15:%.*]] = icmp eq i64 [[TMP14]], 20
+; CHECK-NEXT: br i1 [[TMP15]], label %[[EXIT:.*]], label %[[SCALAR_PH1]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[TMP14]], %[[MIDDLE_BLOCK]] ], [ 0, %[[SCALAR_PH]] ]
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[LD_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[LOAD_ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[LD_ADDR1]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
-; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -344,22 +409,22 @@ exit:
define void @loop_contains_store_unknown_bounds(ptr align 2 dereferenceable(100) noalias %array, ptr align 2 dereferenceable(100) readonly %pred, i64 %n) {
; CHECK-LABEL: define void @loop_contains_store_unknown_bounds(
; CHECK-SAME: ptr noalias align 2 dereferenceable(100) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(100) [[PRED:%.*]], i64 [[N:%.*]]) #[[ATTR0]] {
-; CHECK-NEXT: [[ENTRY:.*]]:
-; CHECK-NEXT: br label %[[FOR_BODY:.*]]
-; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR1]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -435,21 +500,53 @@ define void @loop_contains_store_in_latch_block(ptr dereferenceable(40) noalias
; CHECK-LABEL: define void @loop_contains_store_in_latch_block(
; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 20, [[TMP1]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH1:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 20, [[TMP3]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 20, [[N_MOD_VF]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[FOR_BODY]] ]
; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
-; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 8 x i16>, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[TMP5:%.*]] = icmp sgt <vscale x 8 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP6:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP5]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 0, i64 [[TMP6]])
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i16> @llvm.masked.load.nxv8i16.p0(ptr align 2 [[TMP7]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]], <vscale x 8 x i16> poison)
+; CHECK-NEXT: [[TMP8:%.*]] = add nsw <vscale x 8 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.nxv8i16.p0(<vscale x 8 x i16> [[TMP8]], ptr align 2 [[TMP7]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP9:%.*]] = freeze <vscale x 8 x i1> [[TMP5]]
+; CHECK-NEXT: [[TMP10:%.*]] = call i1 @llvm.vector.reduce.or.nxv8i1(<vscale x 8 x i1> [[TMP9]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], [[TMP3]]
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: [[TMP12:%.*]] = or i1 [[TMP10]], [[TMP11]]
+; CHECK-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[FOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP13:%.*]] = add i64 [[IV]], [[TMP6]]
+; CHECK-NEXT: [[TMP14:%.*]] = icmp eq i64 [[TMP13]], 20
+; CHECK-NEXT: br i1 [[TMP14]], label %[[EXIT:.*]], label %[[SCALAR_PH1]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[TMP13]], %[[MIDDLE_BLOCK]] ], [ 0, %[[SCALAR_PH]] ]
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[EE_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR1]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
-; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP7:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -1076,21 +1173,53 @@ define i32 @uncountable_exit_with_separate_exit_block(ptr dereferenceable(40) no
; CHECK-LABEL: define i32 @uncountable_exit_with_separate_exit_block(
; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) #[[ATTR0]] {
; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[TMP0:%.*]] = call i64 @llvm.vscale.i64()
+; CHECK-NEXT: [[TMP1:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 20, [[TMP1]]
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP3:%.*]] = shl nuw i64 [[TMP0]], 3
+; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i64 20, [[TMP3]]
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i64 20, [[N_MOD_VF]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[TMP5:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <vscale x 8 x i16>, ptr [[TMP5]], align 2
+; CHECK-NEXT: [[TMP6:%.*]] = icmp sgt <vscale x 8 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP7:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.nxv8i1(<vscale x 8 x i1> [[TMP6]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <vscale x 8 x i1> @llvm.get.active.lane.mask.nxv8i1.i64(i64 0, i64 [[TMP7]])
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <vscale x 8 x i16> @llvm.masked.load.nxv8i16.p0(ptr align 2 [[ST_ADDR]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]], <vscale x 8 x i16> poison)
+; CHECK-NEXT: [[TMP8:%.*]] = add nsw <vscale x 8 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.nxv8i16.p0(<vscale x 8 x i16> [[TMP8]], ptr align 2 [[ST_ADDR]], <vscale x 8 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP9:%.*]] = freeze <vscale x 8 x i1> [[TMP6]]
+; CHECK-NEXT: [[TMP10:%.*]] = call i1 @llvm.vector.reduce.or.nxv8i1(<vscale x 8 x i1> [[TMP9]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], [[TMP3]]
+; CHECK-NEXT: [[TMP11:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: [[TMP12:%.*]] = or i1 [[TMP10]], [[TMP11]]
+; CHECK-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[FOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP13:%.*]] = add i64 [[IV]], [[TMP7]]
+; CHECK-NEXT: [[TMP14:%.*]] = icmp eq i64 [[TMP13]], 20
+; CHECK-NEXT: br i1 [[TMP14]], label %[[EXIT_COUNTABLE:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i64 [ [[TMP13]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR1]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT_UNCOUNTABLE:.*]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT_COUNTABLE:.*]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT_COUNTABLE]], label %[[FOR_BODY1]], !llvm.loop [[LOOP9:![0-9]+]]
; CHECK: [[EXIT_COUNTABLE]]:
; CHECK-NEXT: ret i32 0
; CHECK: [[EXIT_UNCOUNTABLE]]:
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll b/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll
index a48f7b30ca573..00ef6968159d1 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/early_exit_with_stores_vplan.ll
@@ -1,12 +1,73 @@
; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
-; RUN: opt -p loop-vectorize -force-vector-width=4 -debug-only=loop-vectorize -disable-output -vplan-print-after=printOptimizedVPlan %s 2>&1 | FileCheck %s
+; RUN: opt -p loop-vectorize -force-vector-width=4 -disable-output -vplan-print-after=printOptimizedVPlan -enable-early-exit-vectorization-with-side-effects -force-target-supports-masked-memory-ops < %s 2>&1 | FileCheck %s
; REQUIRES: asserts
define void @loop_contains_store_condition_load_has_single_user(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
-; CHECK-LABEL: LV: Checking a loop in 'loop_contains_store_condition_load_has_single_user'
-; CHECK: LV: Found an early exit loop
-; CHECK-NEXT: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK-LABEL: VPlan for loop in 'loop_contains_store_condition_load_has_single_user'
+; CHECK: VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT: Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT: Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT: Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT: Live-in ir<20> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<entry>:
+; CHECK-NEXT: Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.ph:
+; CHECK-NEXT: Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION nuw nsw ir<0>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT: CLONE ir<%st.addr> = getelementptr ir<%array>, vp<[[VP4]]>
+; CHECK-NEXT: CLONE ir<%ee.addr> = getelementptr inbounds nuw ir<%pred>, vp<[[VP4]]>
+; CHECK-NEXT: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds nuw ir<%ee.addr>, ir<1>
+; CHECK-NEXT: WIDEN ir<%ee.val> = load vp<[[VP5]]>
+; CHECK-NEXT: WIDEN ir<%ee.cond> = icmp sgt ir<%ee.val>, ir<500>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = first-active-lane ir<%ee.cond>
+; CHECK-NEXT: EMIT vp<%uncountable.exit.mask> = active lane mask ir<0>, vp<[[VP6]]>, ir<1>
+; CHECK-NEXT: vp<[[VP7:%[0-9]+]]> = vector-pointer ir<%st.addr>, ir<1>
+; CHECK-NEXT: WIDEN ir<%data> = load vp<[[VP7]]>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: WIDEN ir<%inc> = add nsw ir<%data>, ir<1>
+; CHECK-NEXT: vp<[[VP8:%[0-9]+]]> = vector-pointer ir<%st.addr>, ir<1>
+; CHECK-NEXT: WIDEN store vp<[[VP8]]>, ir<%inc>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = any-of ir<%ee.cond>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT: EMIT branch-on-two-conds vp<[[VP9]]>, vp<[[VP10]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT: middle.block:
+; CHECK-NEXT: EMIT vp<[[VP12:%[0-9]+]]> = extract-lane ir<0>, ir<%iv>
+; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = add vp<[[VP12]]>, vp<[[VP6]]>
+; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = icmp eq vp<[[VP13]]>, ir<20>
+; CHECK-NEXT: EMIT branch-on-cond vp<[[VP14]]>
+; CHECK-NEXT: Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<exit>:
+; CHECK-NEXT: No successors
+; CHECK-EMPTY:
+; CHECK-NEXT: scalar.ph:
+; CHECK-NEXT: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP13]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT: Successor(s): ir-bb<for.body>
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<for.body>:
+; CHECK-NEXT: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.inc ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT: IR %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+; CHECK-NEXT: IR %data = load i16, ptr %st.addr, align 2
+; CHECK-NEXT: IR %inc = add nsw i16 %data, 1
+; CHECK-NEXT: IR store i16 %inc, ptr %st.addr, align 2
+; CHECK-NEXT: IR %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+; CHECK-NEXT: IR %ee.val = load i16, ptr %ee.addr, align 2
+; CHECK-NEXT: IR %ee.cond = icmp sgt i16 %ee.val, 500
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
;
entry:
br label %for.body
@@ -31,10 +92,67 @@ exit:
ret void
}
-define void @loop_contains_store_after_uncountable_exit(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
-; CHECK-LABEL: LV: Checking a loop in 'loop_contains_store_after_uncountable_exit'
-; CHECK: LV: Found an early exit loop
-; CHECK-NEXT: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+define void @loop_contains_store_after_uncountable_exit(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {;
+; CHECK-LABEL: VPlan for loop in 'loop_contains_store_after_uncountable_exit'
+; CHECK: VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT: Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT: Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT: Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT: Live-in ir<20> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<entry>:
+; CHECK-NEXT: Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.ph:
+; CHECK-NEXT: Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION nuw nsw ir<0>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT: CLONE ir<%ee.addr> = getelementptr inbounds nuw ir<%pred>, vp<[[VP4]]>
+; CHECK-NEXT: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds nuw ir<%ee.addr>, ir<1>
+; CHECK-NEXT: WIDEN ir<%ee.val> = load vp<[[VP5]]>
+; CHECK-NEXT: WIDEN ir<%ee.cond> = icmp sgt ir<%ee.val>, ir<500>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = first-active-lane ir<%ee.cond>
+; CHECK-NEXT: EMIT vp<%uncountable.exit.mask> = active lane mask ir<0>, vp<[[VP6]]>, ir<1>
+; CHECK-NEXT: CLONE ir<%st.addr> = getelementptr ir<%array>, vp<[[VP4]]>
+; CHECK-NEXT: vp<[[VP7:%[0-9]+]]> = vector-pointer ir<%st.addr>, ir<1>
+; CHECK-NEXT: WIDEN ir<%data> = load vp<[[VP7]]>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: WIDEN ir<%inc> = add nsw ir<%data>, ir<1>
+; CHECK-NEXT: vp<[[VP8:%[0-9]+]]> = vector-pointer ir<%st.addr>, ir<1>
+; CHECK-NEXT: WIDEN store vp<[[VP8]]>, ir<%inc>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = any-of ir<%ee.cond>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT: EMIT branch-on-two-conds vp<[[VP9]]>, vp<[[VP10]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT: middle.block:
+; CHECK-NEXT: EMIT vp<[[VP12:%[0-9]+]]> = extract-lane ir<0>, ir<%iv>
+; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = add vp<[[VP12]]>, vp<[[VP6]]>
+; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = icmp eq vp<[[VP13]]>, ir<20>
+; CHECK-NEXT: EMIT branch-on-cond vp<[[VP14]]>
+; CHECK-NEXT: Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<exit>:
+; CHECK-NEXT: No successors
+; CHECK-EMPTY:
+; CHECK-NEXT: scalar.ph:
+; CHECK-NEXT: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP13]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT: Successor(s): ir-bb<for.body>
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<for.body>:
+; CHECK-NEXT: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.inc ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT: IR %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+; CHECK-NEXT: IR %ee.val = load i16, ptr %ee.addr, align 2
+; CHECK-NEXT: IR %ee.cond = icmp sgt i16 %ee.val, 500
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
;
entry:
br label %for.body
@@ -60,9 +178,73 @@ exit:
}
define i16 @uncountable_exit_with_live_out(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
-; CHECK-LABEL: LV: Checking a loop in 'uncountable_exit_with_live_out'
-; CHECK: LV: Found an early exit loop
-; CHECK-NEXT: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK-LABEL: VPlan for loop in 'uncountable_exit_with_live_out'
+; CHECK: VPlan 'Initial VPlan for VF={4},UF>=1' {
+; CHECK-NEXT: Live-in vp<[[VP0:%[0-9]+]]> = VF
+; CHECK-NEXT: Live-in vp<[[VP1:%[0-9]+]]> = VF * UF
+; CHECK-NEXT: Live-in vp<[[VP2:%[0-9]+]]> = vector-trip-count
+; CHECK-NEXT: Live-in ir<20> = original trip-count
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<entry>:
+; CHECK-NEXT: Successor(s): scalar.ph, vector.ph
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.ph:
+; CHECK-NEXT: Successor(s): vector loop
+; CHECK-EMPTY:
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION nuw nsw ir<0>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT: vp<[[VP4:%[0-9]+]]> = SCALAR-STEPS vp<[[VP3]]>, ir<1>, vp<[[VP0]]>
+; CHECK-NEXT: CLONE ir<%st.addr> = getelementptr ir<%array>, vp<[[VP4]]>
+; CHECK-NEXT: CLONE ir<%ee.addr> = getelementptr inbounds nuw ir<%pred>, vp<[[VP4]]>
+; CHECK-NEXT: vp<[[VP5:%[0-9]+]]> = vector-pointer inbounds nuw ir<%ee.addr>, ir<1>
+; CHECK-NEXT: WIDEN ir<%ee.val> = load vp<[[VP5]]>
+; CHECK-NEXT: WIDEN ir<%ee.cond> = icmp sgt ir<%ee.val>, ir<500>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = first-active-lane ir<%ee.cond>
+; CHECK-NEXT: EMIT vp<%uncountable.exit.mask> = active lane mask ir<0>, vp<[[VP6]]>, ir<1>
+; CHECK-NEXT: vp<[[VP7:%[0-9]+]]> = vector-pointer ir<%st.addr>, ir<1>
+; CHECK-NEXT: WIDEN ir<%data> = load vp<[[VP7]]>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: WIDEN ir<%inc> = add nsw ir<%data>, ir<1>
+; CHECK-NEXT: vp<[[VP8:%[0-9]+]]> = vector-pointer ir<%st.addr>, ir<1>
+; CHECK-NEXT: WIDEN store vp<[[VP8]]>, ir<%inc>, vp<%uncountable.exit.mask>
+; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = any-of ir<%ee.cond>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1]]>
+; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = icmp eq vp<%index.next>, vp<[[VP2]]>
+; CHECK-NEXT: EMIT branch-on-two-conds vp<[[VP9]]>, vp<[[VP10]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block, middle.block
+; CHECK-EMPTY:
+; CHECK-NEXT: middle.block:
+; CHECK-NEXT: EMIT vp<[[VP12:%[0-9]+]]> = extract-last-part ir<%data>
+; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = extract-last-lane vp<[[VP12]]>
+; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = extract-lane ir<0>, ir<%iv>
+; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = add vp<[[VP14]]>, vp<[[VP6]]>
+; CHECK-NEXT: EMIT vp<[[VP16:%[0-9]+]]> = icmp eq vp<[[VP15]]>, ir<20>
+; CHECK-NEXT: EMIT branch-on-cond vp<[[VP16]]>
+; CHECK-NEXT: Successor(s): ir-bb<exit>, scalar.ph
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<exit>:
+; CHECK-NEXT: IR %data.lcssa = phi i16 [ %data, %for.inc ], [ %data, %for.body ] (extra operand: vp<[[VP13]]> from middle.block)
+; CHECK-NEXT: No successors
+; CHECK-EMPTY:
+; CHECK-NEXT: scalar.ph:
+; CHECK-NEXT: EMIT-SCALAR vp<%bc.resume.val> = phi [ vp<[[VP15]]>, middle.block ], [ ir<0>, ir-bb<entry> ]
+; CHECK-NEXT: Successor(s): ir-bb<for.body>
+; CHECK-EMPTY:
+; CHECK-NEXT: ir-bb<for.body>:
+; CHECK-NEXT: IR %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.inc ] (extra operand: vp<%bc.resume.val> from scalar.ph)
+; CHECK-NEXT: IR %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+; CHECK-NEXT: IR %data = load i16, ptr %st.addr, align 2
+; CHECK-NEXT: IR %inc = add nsw i16 %data, 1
+; CHECK-NEXT: IR store i16 %inc, ptr %st.addr, align 2
+; CHECK-NEXT: IR %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+; CHECK-NEXT: IR %ee.val = load i16, ptr %ee.addr, align 2
+; CHECK-NEXT: IR %ee.cond = icmp sgt i16 %ee.val, 500
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
;
entry:
br label %for.body
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll b/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll
index a46934f7b4a07..12065c01d47a4 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_store_legality.ll
@@ -1,5 +1,5 @@
; REQUIRES: asserts
-; RUN: opt -S < %s -p loop-vectorize -debug-only=loop-vectorize -force-vector-width=4 -disable-output 2>&1 | FileCheck %s
+; RUN: opt -S < %s -p loop-vectorize -debug-only=loop-vectorize -enable-early-exit-vectorization-with-side-effects -force-vector-width=4 -disable-output 2>&1 | FileCheck %s
;; This currently doesn't vectorize because the load used to determine the
;; uncountable exit condition has a second user (the store).
@@ -30,7 +30,7 @@ loop.end:
define void @loop_contains_store_condition_load_has_single_user(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: LV: Checking a loop in 'loop_contains_store_condition_load_has_single_user'
-; CHECK: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK: LV: We can vectorize this loop!
entry:
br label %for.body
@@ -176,7 +176,8 @@ exit:
;; Alternatively, we could use masked.load.ff or vp.load.ff
define void @loop_contains_store_assumed_bounds(ptr noalias %array, ptr readonly %pred, i64 %n) {
; CHECK-LABEL: LV: Checking a loop in 'loop_contains_store_assumed_bounds'
-; CHECK: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK: LV: We can vectorize this loop!
+; CHECK: LV: Vectorization is possible but not beneficial.
entry:
%n_bytes = mul nuw nsw i64 %n, 2
call void @llvm.assume(i1 true) [ "align"(ptr %pred, i64 2), "dereferenceable"(ptr %pred, i64 %n_bytes) ]
@@ -204,7 +205,7 @@ exit:
define void @loop_contains_store_to_pointer_with_no_deref_info(ptr align 2 dereferenceable(40) readonly %load.array, ptr align 2 noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: LV: Checking a loop in 'loop_contains_store_to_pointer_with_no_deref_info'
-; CHECK: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK: LV: We can vectorize this loop!
entry:
br label %for.body
@@ -232,7 +233,8 @@ exit:
;; Vectorizeable, requires runtime checks and/or ff loads.
define void @loop_contains_store_unknown_bounds(ptr align 2 dereferenceable(100) noalias %array, ptr align 2 dereferenceable(100) readonly %pred, i64 %n) {
; CHECK-LABEL: LV: Checking a loop in 'loop_contains_store_unknown_bounds'
-; CHECK: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK: LV: We can vectorize this loop!
+; CHECK: LV: Vectorization is possible but not beneficial.
entry:
br label %for.body
@@ -287,7 +289,7 @@ exit:
;; Vectorizeable, but we really want LICM to sink the store out of the loop
define void @loop_contains_store_to_invariant_location(ptr dereferenceable(40) readonly %array, ptr align 2 dereferenceable(40) readonly %pred, ptr noalias %store_addr) {
; CHECK-LABEL: LV: Checking a loop in 'loop_contains_store_to_invariant_location'
-; CHECK: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK: LV: Not vectorizing: Cannot vectorize early exit loops with stores to loop-invariant addresses.
entry:
br label %for.body
@@ -313,7 +315,7 @@ exit:
define void @loop_contains_store_in_latch_block(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: LV: Checking a loop in 'loop_contains_store_in_latch_block'
-; CHECK: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK: LV: We can vectorize this loop!
entry:
br label %for.body
@@ -366,7 +368,7 @@ exit:
define void @loop_contains_store_decrementing_iv(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: LV: Checking a loop in 'loop_contains_store_decrementing_iv'
-; CHECK: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK: LV: We can vectorize this loop!
entry:
br label %for.body
@@ -665,7 +667,7 @@ exit:
define i16 @uncountable_exit_with_live_out(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: LV: Checking a loop in 'uncountable_exit_with_live_out'
-; CHECK: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK: LV: We can vectorize this loop!
entry:
br label %for.body
@@ -692,7 +694,8 @@ exit:
; Vectorizeable, requires improvements in dereferenceability checks
define void @uncountable_exit_with_constant_nonunit_stride(ptr dereferenceable(4000) noalias %array, ptr align 2 dereferenceable(4000) readonly %pred) {
; CHECK-LABEL: LV: Checking a loop in 'uncountable_exit_with_constant_nonunit_stride'
-; CHECK: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK: LV: We can vectorize this loop!
+; CHECK: LV: Not vectorizing: unable to calculate the loop count due to complex control flow.
entry:
br label %for.body
@@ -745,7 +748,7 @@ exit:
define i32 @uncountable_exit_with_separate_exit_block(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: LV: Checking a loop in 'uncountable_exit_with_separate_exit_block'
-; CHECK: LV: Not vectorizing: Writes to memory unsupported in early exit loops.
+; CHECK: LV: We can vectorize this loop!
entry:
br label %for.body
@@ -772,6 +775,45 @@ exit.uncountable:
ret i32 1
}
+define i32 @uncountable_exit_with_masked_ldst_separate_condition(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred, ptr align 2 readonly %st.pred) {
+; CHECK-LABEL: LV: Checking a loop in 'uncountable_exit_with_masked_ldst_separate_condition'
+; CHECK: LV: We can vectorize this loop!
+; CHECK: LV: Vectorization is possible but not beneficial.
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.inc ]
+ %stp.gep = getelementptr inbounds nuw i16, ptr %st.pred, i64 %iv
+ %stp.val = load i16, ptr %stp.gep, align 2
+ %stp.cond = icmp slt i16 %stp.val, 2345
+ br i1 %stp.cond, label %ldst.block, label %ee.block
+
+ldst.block:
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ br label %ee.block
+
+ee.block:
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit.uncountable, label %for.inc
+
+for.inc:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit.countable, label %for.body
+
+exit.countable:
+ ret i32 0
+
+exit.uncountable:
+ ret i32 1
+}
+
;; Avoid vectorization; similar to another invariant test above, we would either
;; exit immediately on the first lane or never take the early exit. Should be
;; versioned before reaching LV.
diff --git a/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll b/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll
index ed495b33e1fbd..affe5e30c1146 100644
--- a/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll
+++ b/llvm/test/Transforms/LoopVectorize/early_exit_with_stores.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 | FileCheck %s
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 -force-target-supports-masked-memory-ops -enable-early-exit-vectorization-with-side-effects | FileCheck %s
;; See early_exit_store_legality.ll for reasons why a particular loop doesn't
;; vectorize yet.
@@ -50,22 +50,47 @@ loop.end:
define void @loop_contains_store_condition_load_has_single_user(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: define void @loop_contains_store_condition_load_has_single_user(
; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
-; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[TMP3:%.*]] = phi i64 [ 0, %[[FOR_BODY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[TMP3]]
+; CHECK-NEXT: [[TMP12:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[TMP3]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP12]], align 2
+; CHECK-NEXT: [[TMP13:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP14:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP13]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP14]])
+; CHECK-NEXT: [[TMP30:%.*]] = call <4 x i16> @llvm.masked.load.v4i16.p0(ptr align 2 [[TMP7]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i16> poison)
+; CHECK-NEXT: [[TMP31:%.*]] = add nsw <4 x i16> [[TMP30]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.v4i16.p0(<4 x i16> [[TMP31]], ptr align 2 [[TMP7]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP40:%.*]] = freeze <4 x i1> [[TMP13]]
+; CHECK-NEXT: [[TMP41:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP40]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[TMP3]], 4
+; CHECK-NEXT: [[TMP42:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP43:%.*]] = or i1 [[TMP41]], [[TMP42]]
+; CHECK-NEXT: br i1 [[TMP43]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP44:%.*]] = add i64 [[TMP3]], [[TMP14]]
+; CHECK-NEXT: [[TMP45:%.*]] = icmp eq i64 [[TMP44]], 20
+; CHECK-NEXT: br i1 [[TMP45]], label %[[EXIT:.*]], label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[TMP44]], %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA1:%.*]] = load i16, ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[INC1:%.*]] = add nsw i16 [[DATA1]], 1
+; CHECK-NEXT: store i16 [[INC1]], ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
-; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -251,21 +276,21 @@ define void @loop_contains_store_assumed_bounds(ptr noalias %array, ptr readonly
; CHECK-NEXT: [[ENTRY:.*]]:
; CHECK-NEXT: [[N_BYTES:%.*]] = mul nuw nsw i64 [[N]], 2
; CHECK-NEXT: call void @llvm.assume(i1 true) [ "align"(ptr [[PRED]], i64 2), "dereferenceable"(ptr [[PRED]], i64 [[N_BYTES]]) ]
-; CHECK-NEXT: br label %[[FOR_BODY:.*]]
-; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA1:%.*]] = load i16, ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[INC1:%.*]] = add nsw i16 [[DATA1]], 1
+; CHECK-NEXT: store i16 [[INC1]], ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -297,23 +322,49 @@ exit:
define void @loop_contains_store_to_pointer_with_no_deref_info(ptr align 2 dereferenceable(40) readonly %load.array, ptr align 2 noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: define void @loop_contains_store_to_pointer_with_no_deref_info(
; CHECK-SAME: ptr readonly align 2 dereferenceable(40) [[LOAD_ARRAY:%.*]], ptr noalias align 2 [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
-; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[LD_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[LOAD_ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[LD_ADDR]], align 2
-; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[FOR_BODY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr i16, ptr [[LOAD_ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP4]], align 2
+; CHECK-NEXT: [[TMP5:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP6:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP5]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP6]])
+; CHECK-NEXT: [[TMP26:%.*]] = call <4 x i16> @llvm.masked.load.v4i16.p0(ptr align 2 [[TMP0]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i16> poison)
+; CHECK-NEXT: [[TMP27:%.*]] = add nsw <4 x i16> [[TMP26]], splat (i16 1)
+; CHECK-NEXT: [[TMP29:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: call void @llvm.masked.store.v4i16.p0(<4 x i16> [[TMP27]], ptr align 2 [[TMP29]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP40:%.*]] = freeze <4 x i1> [[TMP5]]
+; CHECK-NEXT: [[TMP41:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP40]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP42:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP43:%.*]] = or i1 [[TMP41]], [[TMP42]]
+; CHECK-NEXT: br i1 [[TMP43]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP44:%.*]] = add i64 [[INDEX]], [[TMP6]]
+; CHECK-NEXT: [[TMP45:%.*]] = icmp eq i64 [[TMP44]], 20
+; CHECK-NEXT: br i1 [[TMP45]], label %[[EXIT:.*]], label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[TMP44]], %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[LD_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[LOAD_ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA1:%.*]] = load i16, ptr [[LD_ADDR1]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA1]], 1
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
-; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -344,22 +395,22 @@ exit:
define void @loop_contains_store_unknown_bounds(ptr align 2 dereferenceable(100) noalias %array, ptr align 2 dereferenceable(100) readonly %pred, i64 %n) {
; CHECK-LABEL: define void @loop_contains_store_unknown_bounds(
; CHECK-SAME: ptr noalias align 2 dereferenceable(100) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(100) [[PRED:%.*]], i64 [[N:%.*]]) {
-; CHECK-NEXT: [[ENTRY:.*]]:
-; CHECK-NEXT: br label %[[FOR_BODY:.*]]
-; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA1:%.*]] = load i16, ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[INC1:%.*]] = add nsw i16 [[DATA1]], 1
+; CHECK-NEXT: store i16 [[INC1]], ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], [[N]]
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -389,22 +440,47 @@ exit:
define void @loop_contains_store_in_latch_block(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: define void @loop_contains_store_in_latch_block(
; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
-; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: [[SCALAR_PH:.*:]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
-; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[FOR_BODY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[TMP5:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP6:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP5]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP6]])
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP30:%.*]] = call <4 x i16> @llvm.masked.load.v4i16.p0(ptr align 2 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i16> poison)
+; CHECK-NEXT: [[TMP31:%.*]] = add nsw <4 x i16> [[TMP30]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.v4i16.p0(<4 x i16> [[TMP31]], ptr align 2 [[TMP3]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP40:%.*]] = freeze <4 x i1> [[TMP5]]
+; CHECK-NEXT: [[TMP41:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP40]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP42:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP43:%.*]] = or i1 [[TMP41]], [[TMP42]]
+; CHECK-NEXT: br i1 [[TMP43]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP44:%.*]] = add i64 [[INDEX]], [[TMP6]]
+; CHECK-NEXT: [[TMP45:%.*]] = icmp eq i64 [[TMP44]], 20
+; CHECK-NEXT: br i1 [[TMP45]], label %[[EXIT:.*]], label %[[SCALAR_PH1:.*]]
+; CHECK: [[SCALAR_PH1]]:
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[TMP44]], %[[SCALAR_PH1]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[EE_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR1]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
-; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP7:![0-9]+]]
; CHECK: [[EXIT]]:
; CHECK-NEXT: ret void
;
@@ -940,24 +1016,50 @@ exit:
define i16 @uncountable_exit_with_live_out(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: define i16 @uncountable_exit_with_live_out(
; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
-; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
-; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
-; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
-; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
-; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[TMP3:%.*]] = phi i64 [ 0, %[[FOR_BODY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[TMP3]]
+; CHECK-NEXT: [[TMP12:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[TMP3]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP12]], align 2
+; CHECK-NEXT: [[TMP13:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP14:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP13]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP14]])
+; CHECK-NEXT: [[TMP30:%.*]] = call <4 x i16> @llvm.masked.load.v4i16.p0(ptr align 2 [[TMP7]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i16> poison)
+; CHECK-NEXT: [[TMP31:%.*]] = add nsw <4 x i16> [[TMP30]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.v4i16.p0(<4 x i16> [[TMP31]], ptr align 2 [[TMP7]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP40:%.*]] = freeze <4 x i1> [[TMP13]]
+; CHECK-NEXT: [[TMP41:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP40]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[TMP3]], 4
+; CHECK-NEXT: [[TMP42:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP43:%.*]] = or i1 [[TMP41]], [[TMP42]]
+; CHECK-NEXT: br i1 [[TMP43]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP9:%.*]] = extractelement <4 x i16> [[TMP30]], i64 3
+; CHECK-NEXT: [[TMP45:%.*]] = add i64 [[TMP3]], [[TMP14]]
+; CHECK-NEXT: [[TMP46:%.*]] = icmp eq i64 [[TMP45]], 20
+; CHECK-NEXT: br i1 [[TMP46]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[TMP45]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA1:%.*]] = load i16, ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[INC1:%.*]] = add nsw i16 [[DATA1]], 1
+; CHECK-NEXT: store i16 [[INC1]], ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
-; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT:.*]], label %[[FOR_INC]]
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
; CHECK: [[FOR_INC]]:
-; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
-; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]]
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY1]], !llvm.loop [[LOOP9:![0-9]+]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[DATA_LCSSA:%.*]] = phi i16 [ [[DATA]], %[[FOR_INC]] ], [ [[DATA]], %[[FOR_BODY]] ]
+; CHECK-NEXT: [[DATA_LCSSA:%.*]] = phi i16 [ [[DATA1]], %[[FOR_INC]] ], [ [[DATA1]], %[[FOR_BODY1]] ], [ [[TMP9]], %[[MIDDLE_BLOCK]] ]
; CHECK-NEXT: ret i16 [[DATA_LCSSA]]
;
entry:
@@ -1076,14 +1178,96 @@ exit:
define i32 @uncountable_exit_with_separate_exit_block(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
; CHECK-LABEL: define i32 @uncountable_exit_with_separate_exit_block(
; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
-; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: br label %[[FOR_BODY:.*]]
; CHECK: [[FOR_BODY]]:
-; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[TMP3:%.*]] = phi i64 [ 0, %[[FOR_BODY]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP7:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[TMP3]]
+; CHECK-NEXT: [[TMP12:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[TMP3]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP12]], align 2
+; CHECK-NEXT: [[TMP13:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP14:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP13]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP14]])
+; CHECK-NEXT: [[TMP30:%.*]] = call <4 x i16> @llvm.masked.load.v4i16.p0(ptr align 2 [[TMP7]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i16> poison)
+; CHECK-NEXT: [[TMP31:%.*]] = add nsw <4 x i16> [[TMP30]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.v4i16.p0(<4 x i16> [[TMP31]], ptr align 2 [[TMP7]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP40:%.*]] = freeze <4 x i1> [[TMP13]]
+; CHECK-NEXT: [[TMP41:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP40]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[TMP3]], 4
+; CHECK-NEXT: [[TMP42:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP43:%.*]] = or i1 [[TMP41]], [[TMP42]]
+; CHECK-NEXT: br i1 [[TMP43]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP44:%.*]] = add i64 [[TMP3]], [[TMP14]]
+; CHECK-NEXT: [[TMP45:%.*]] = icmp eq i64 [[TMP44]], 20
+; CHECK-NEXT: br i1 [[TMP45]], label %[[EXIT_COUNTABLE:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[FOR_BODY1:.*]]
+; CHECK: [[FOR_BODY1]]:
+; CHECK-NEXT: [[IV1:%.*]] = phi i64 [ [[TMP44]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR1:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV1]]
+; CHECK-NEXT: [[DATA1:%.*]] = load i16, ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[INC1:%.*]] = add nsw i16 [[DATA1]], 1
+; CHECK-NEXT: store i16 [[INC1]], ptr [[ST_ADDR1]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV1]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT_UNCOUNTABLE:.*]], label %[[FOR_INC]]
+; CHECK: [[FOR_INC]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV1]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT_COUNTABLE]], label %[[FOR_BODY1]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK: [[EXIT_COUNTABLE]]:
+; CHECK-NEXT: ret i32 0
+; CHECK: [[EXIT_UNCOUNTABLE]]:
+; CHECK-NEXT: ret i32 1
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.inc ]
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit.uncountable, label %for.inc
+
+for.inc:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit.countable, label %for.body
+
+exit.countable:
+ ret i32 0
+
+exit.uncountable:
+ ret i32 1
+}
+
+define i32 @uncountable_exit_with_masked_ldst_separate_condition(ptr noalias %array, ptr align 2 dereferenceable(40) readonly %pred, ptr align 2 readonly %st.pred) {
+; CHECK-LABEL: define i32 @uncountable_exit_with_masked_ldst_separate_condition(
+; CHECK-SAME: ptr noalias [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]], ptr readonly align 2 [[ST_PRED:%.*]]) {
+; CHECK-NEXT: [[SCALAR_PH:.*]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[STP_GEP:%.*]] = getelementptr inbounds nuw i16, ptr [[ST_PRED]], i64 [[IV]]
+; CHECK-NEXT: [[STP_VAL:%.*]] = load i16, ptr [[STP_GEP]], align 2
+; CHECK-NEXT: [[STP_COND:%.*]] = icmp slt i16 [[STP_VAL]], 2345
+; CHECK-NEXT: br i1 [[STP_COND]], label %[[LDST_BLOCK:.*]], label %[[EE_BLOCK:.*]]
+; CHECK: [[LDST_BLOCK]]:
; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: br label %[[EE_BLOCK]]
+; CHECK: [[EE_BLOCK]]:
; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
@@ -1102,10 +1286,19 @@ entry:
for.body:
%iv = phi i64 [ 0, %entry ], [ %iv.next, %for.inc ]
+ %stp.gep = getelementptr inbounds nuw i16, ptr %st.pred, i64 %iv
+ %stp.val = load i16, ptr %stp.gep, align 2
+ %stp.cond = icmp slt i16 %stp.val, 2345
+ br i1 %stp.cond, label %ldst.block, label %ee.block
+
+ldst.block:
%st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
%data = load i16, ptr %st.addr, align 2
%inc = add nsw i16 %data, 1
store i16 %inc, ptr %st.addr, align 2
+ br label %ee.block
+
+ee.block:
%ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
%ee.val = load i16, ptr %ee.addr, align 2
%ee.cond = icmp sgt i16 %ee.val, 500
diff --git a/llvm/test/Transforms/LoopVectorize/interleave_uncountable_exits.ll b/llvm/test/Transforms/LoopVectorize/interleave_uncountable_exits.ll
new file mode 100644
index 0000000000000..26f4ee4b6ead8
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/interleave_uncountable_exits.ll
@@ -0,0 +1,74 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 -force-vector-interleave=2 -force-target-max-vector-interleave=2 -force-target-supports-masked-memory-ops -enable-early-exit-vectorization-with-side-effects | FileCheck %s
+
+; We do not support interleaving loops with early exits yet, so make sure we don't
+; even with a flag to force it.
+define void @loop_contains_store_condition_load_has_single_user(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
+; CHECK-LABEL: define void @loop_contains_store_condition_load_has_single_user(
+; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP1]], align 2
+; CHECK-NEXT: [[TMP2:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP3:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP2]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP3]])
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x i16> @llvm.masked.load.v4i16.p0(ptr align 2 [[TMP0]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i16> poison)
+; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.v4i16.p0(<4 x i16> [[TMP4]], ptr align 2 [[TMP0]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP5:%.*]] = freeze <4 x i1> [[TMP2]]
+; CHECK-NEXT: [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], [[TMP3]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[TMP9]], 20
+; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP9]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
+; CHECK: [[FOR_INC]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.inc ]
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %for.inc
+
+for.inc:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %for.body
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/scalarized_conditional_ops_uncountable_exits.ll b/llvm/test/Transforms/LoopVectorize/scalarized_conditional_ops_uncountable_exits.ll
new file mode 100644
index 0000000000000..942212dfc63d1
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/scalarized_conditional_ops_uncountable_exits.ll
@@ -0,0 +1,132 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 -enable-early-exit-vectorization-with-side-effects - | FileCheck %s
+
+define void @loop_contains_store_condition_load_has_single_user(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
+; CHECK-LABEL: define void @loop_contains_store_condition_load_has_single_user(
+; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE12:.*]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT: [[TMP25:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT: [[TMP26:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT: [[TMP31:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP32:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP0]]
+; CHECK-NEXT: [[TMP33:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP25]]
+; CHECK-NEXT: [[TMP34:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[TMP26]]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP1]], align 2
+; CHECK-NEXT: [[TMP2:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP3:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP2]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP3]])
+; CHECK-NEXT: [[TMP35:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 0
+; CHECK-NEXT: br i1 [[TMP35]], label %[[PRED_LOAD_IF:.*]], label %[[PRED_LOAD_CONTINUE:.*]]
+; CHECK: [[PRED_LOAD_IF]]:
+; CHECK-NEXT: [[TMP11:%.*]] = load i16, ptr [[TMP31]], align 2
+; CHECK-NEXT: [[TMP12:%.*]] = insertelement <4 x i16> poison, i16 [[TMP11]], i64 0
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE]]
+; CHECK: [[PRED_LOAD_CONTINUE]]:
+; CHECK-NEXT: [[TMP13:%.*]] = phi <4 x i16> [ poison, %[[VECTOR_BODY]] ], [ [[TMP12]], %[[PRED_LOAD_IF]] ]
+; CHECK-NEXT: [[TMP14:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 1
+; CHECK-NEXT: br i1 [[TMP14]], label %[[PRED_LOAD_IF1:.*]], label %[[PRED_LOAD_CONTINUE2:.*]]
+; CHECK: [[PRED_LOAD_IF1]]:
+; CHECK-NEXT: [[TMP15:%.*]] = load i16, ptr [[TMP32]], align 2
+; CHECK-NEXT: [[TMP16:%.*]] = insertelement <4 x i16> [[TMP13]], i16 [[TMP15]], i64 1
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE2]]
+; CHECK: [[PRED_LOAD_CONTINUE2]]:
+; CHECK-NEXT: [[TMP17:%.*]] = phi <4 x i16> [ [[TMP13]], %[[PRED_LOAD_CONTINUE]] ], [ [[TMP16]], %[[PRED_LOAD_IF1]] ]
+; CHECK-NEXT: [[TMP18:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 2
+; CHECK-NEXT: br i1 [[TMP18]], label %[[PRED_LOAD_IF3:.*]], label %[[PRED_LOAD_CONTINUE4:.*]]
+; CHECK: [[PRED_LOAD_IF3]]:
+; CHECK-NEXT: [[TMP19:%.*]] = load i16, ptr [[TMP33]], align 2
+; CHECK-NEXT: [[TMP20:%.*]] = insertelement <4 x i16> [[TMP17]], i16 [[TMP19]], i64 2
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE4]]
+; CHECK: [[PRED_LOAD_CONTINUE4]]:
+; CHECK-NEXT: [[TMP21:%.*]] = phi <4 x i16> [ [[TMP17]], %[[PRED_LOAD_CONTINUE2]] ], [ [[TMP20]], %[[PRED_LOAD_IF3]] ]
+; CHECK-NEXT: [[TMP22:%.*]] = extractelement <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], i64 3
+; CHECK-NEXT: br i1 [[TMP22]], label %[[PRED_LOAD_IF5:.*]], label %[[PRED_LOAD_CONTINUE6:.*]]
+; CHECK: [[PRED_LOAD_IF5]]:
+; CHECK-NEXT: [[TMP23:%.*]] = load i16, ptr [[TMP34]], align 2
+; CHECK-NEXT: [[TMP24:%.*]] = insertelement <4 x i16> [[TMP21]], i16 [[TMP23]], i64 3
+; CHECK-NEXT: br label %[[PRED_LOAD_CONTINUE6]]
+; CHECK: [[PRED_LOAD_CONTINUE6]]:
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = phi <4 x i16> [ [[TMP21]], %[[PRED_LOAD_CONTINUE4]] ], [ [[TMP24]], %[[PRED_LOAD_IF5]] ]
+; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: br i1 [[TMP35]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
+; CHECK: [[PRED_STORE_IF]]:
+; CHECK-NEXT: [[TMP27:%.*]] = extractelement <4 x i16> [[TMP4]], i64 0
+; CHECK-NEXT: store i16 [[TMP27]], ptr [[TMP31]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE]]
+; CHECK: [[PRED_STORE_CONTINUE]]:
+; CHECK-NEXT: br i1 [[TMP14]], label %[[PRED_STORE_IF7:.*]], label %[[PRED_STORE_CONTINUE8:.*]]
+; CHECK: [[PRED_STORE_IF7]]:
+; CHECK-NEXT: [[TMP28:%.*]] = extractelement <4 x i16> [[TMP4]], i64 1
+; CHECK-NEXT: store i16 [[TMP28]], ptr [[TMP32]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE8]]
+; CHECK: [[PRED_STORE_CONTINUE8]]:
+; CHECK-NEXT: br i1 [[TMP18]], label %[[PRED_STORE_IF9:.*]], label %[[PRED_STORE_CONTINUE10:.*]]
+; CHECK: [[PRED_STORE_IF9]]:
+; CHECK-NEXT: [[TMP29:%.*]] = extractelement <4 x i16> [[TMP4]], i64 2
+; CHECK-NEXT: store i16 [[TMP29]], ptr [[TMP33]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE10]]
+; CHECK: [[PRED_STORE_CONTINUE10]]:
+; CHECK-NEXT: br i1 [[TMP22]], label %[[PRED_STORE_IF11:.*]], label %[[PRED_STORE_CONTINUE12]]
+; CHECK: [[PRED_STORE_IF11]]:
+; CHECK-NEXT: [[TMP30:%.*]] = extractelement <4 x i16> [[TMP4]], i64 3
+; CHECK-NEXT: store i16 [[TMP30]], ptr [[TMP34]], align 2
+; CHECK-NEXT: br label %[[PRED_STORE_CONTINUE12]]
+; CHECK: [[PRED_STORE_CONTINUE12]]:
+; CHECK-NEXT: [[TMP5:%.*]] = freeze <4 x i1> [[TMP2]]
+; CHECK-NEXT: [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], [[TMP3]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[TMP9]], 20
+; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP9]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
+; CHECK: [[FOR_INC]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.inc ]
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %for.inc
+
+for.inc:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %for.body
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/tail_fold_uncountable_exits.ll b/llvm/test/Transforms/LoopVectorize/tail_fold_uncountable_exits.ll
new file mode 100644
index 0000000000000..73fef71ce774f
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/tail_fold_uncountable_exits.ll
@@ -0,0 +1,77 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 -force-target-supports-masked-memory-ops -enable-early-exit-vectorization-with-side-effects -tail-folding-policy=must-fold-tail -force-tail-folding-style=data - | FileCheck %s
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 -force-target-supports-masked-memory-ops -enable-early-exit-vectorization-with-side-effects -tail-folding-policy=must-fold-tail -force-tail-folding-style=data-without-lane-mask - | FileCheck %s
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 -force-target-supports-masked-memory-ops -enable-early-exit-vectorization-with-side-effects -tail-folding-policy=must-fold-tail -force-tail-folding-style=data-and-control - | FileCheck %s
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 -force-target-supports-masked-memory-ops -enable-early-exit-vectorization-with-side-effects -tail-folding-policy=must-fold-tail -force-tail-folding-style=data-with-evl - | FileCheck %s
+
+; We do not support interleaving loops with early exits yet, so make sure we don't
+; even with a flag to force it.
+define void @loop_contains_store_condition_load_has_single_user(ptr dereferenceable(40) noalias %array, ptr align 2 dereferenceable(40) readonly %pred) {
+; CHECK-LABEL: define void @loop_contains_store_condition_load_has_single_user(
+; CHECK-SAME: ptr noalias dereferenceable(40) [[ARRAY:%.*]], ptr readonly align 2 dereferenceable(40) [[PRED:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = getelementptr i16, ptr [[ARRAY]], i64 [[INDEX]]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i16>, ptr [[TMP1]], align 2
+; CHECK-NEXT: [[TMP2:%.*]] = icmp sgt <4 x i16> [[WIDE_LOAD]], splat (i16 500)
+; CHECK-NEXT: [[TMP3:%.*]] = call i64 @llvm.experimental.cttz.elts.i64.v4i1(<4 x i1> [[TMP2]], i1 false)
+; CHECK-NEXT: [[UNCOUNTABLE_EXIT_MASK:%.*]] = call <4 x i1> @llvm.get.active.lane.mask.v4i1.i64(i64 0, i64 [[TMP3]])
+; CHECK-NEXT: [[WIDE_MASKED_LOAD:%.*]] = call <4 x i16> @llvm.masked.load.v4i16.p0(ptr align 2 [[TMP0]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]], <4 x i16> poison)
+; CHECK-NEXT: [[TMP4:%.*]] = add nsw <4 x i16> [[WIDE_MASKED_LOAD]], splat (i16 1)
+; CHECK-NEXT: call void @llvm.masked.store.v4i16.p0(<4 x i16> [[TMP4]], ptr align 2 [[TMP0]], <4 x i1> [[UNCOUNTABLE_EXIT_MASK]])
+; CHECK-NEXT: [[TMP5:%.*]] = freeze <4 x i1> [[TMP2]]
+; CHECK-NEXT: [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: [[TMP7:%.*]] = icmp eq i64 [[INDEX_NEXT]], 20
+; CHECK-NEXT: [[TMP8:%.*]] = or i1 [[TMP6]], [[TMP7]]
+; CHECK-NEXT: br i1 [[TMP8]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP9:%.*]] = add i64 [[INDEX]], [[TMP3]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[TMP9]], 20
+; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT:.*]], label %[[SCALAR_PH:.*]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: br label %[[FOR_BODY:.*]]
+; CHECK: [[FOR_BODY]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i64 [ [[TMP9]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[FOR_INC:.*]] ]
+; CHECK-NEXT: [[ST_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[ARRAY]], i64 [[IV]]
+; CHECK-NEXT: [[DATA:%.*]] = load i16, ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[INC:%.*]] = add nsw i16 [[DATA]], 1
+; CHECK-NEXT: store i16 [[INC]], ptr [[ST_ADDR]], align 2
+; CHECK-NEXT: [[EE_ADDR:%.*]] = getelementptr inbounds nuw i16, ptr [[PRED]], i64 [[IV]]
+; CHECK-NEXT: [[EE_VAL:%.*]] = load i16, ptr [[EE_ADDR]], align 2
+; CHECK-NEXT: [[EE_COND:%.*]] = icmp sgt i16 [[EE_VAL]], 500
+; CHECK-NEXT: br i1 [[EE_COND]], label %[[EXIT]], label %[[FOR_INC]]
+; CHECK: [[FOR_INC]]:
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT: [[COUNTED_COND:%.*]] = icmp eq i64 [[IV_NEXT]], 20
+; CHECK-NEXT: br i1 [[COUNTED_COND]], label %[[EXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %for.body
+
+for.body:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %for.inc ]
+ %st.addr = getelementptr inbounds nuw i16, ptr %array, i64 %iv
+ %data = load i16, ptr %st.addr, align 2
+ %inc = add nsw i16 %data, 1
+ store i16 %inc, ptr %st.addr, align 2
+ %ee.addr = getelementptr inbounds nuw i16, ptr %pred, i64 %iv
+ %ee.val = load i16, ptr %ee.addr, align 2
+ %ee.cond = icmp sgt i16 %ee.val, 500
+ br i1 %ee.cond, label %exit, label %for.inc
+
+for.inc:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %counted.cond = icmp eq i64 %iv.next, 20
+ br i1 %counted.cond, label %exit, label %for.body
+
+exit:
+ ret void
+}
diff --git a/llvm/unittests/Transforms/Vectorize/VPlanUncountableExitTest.cpp b/llvm/unittests/Transforms/Vectorize/VPlanUncountableExitTest.cpp
index 5f37c876a0894..3e10d5f17699d 100644
--- a/llvm/unittests/Transforms/Vectorize/VPlanUncountableExitTest.cpp
+++ b/llvm/unittests/Transforms/Vectorize/VPlanUncountableExitTest.cpp
@@ -66,8 +66,8 @@ static void combineExitConditions(VPlan &Plan) {
VPValue *IsLatchExitTaken = LatchBranch->getOperand(0);
LatchBranch->eraseFromParent();
Builder.setInsertPoint(LatchVPBB);
- Builder.createNaryOp(VPInstruction::BranchOnCond,
- {Builder.createOr(IsAnyExitTaken, IsLatchExitTaken)});
+ Builder.createNaryOp(VPInstruction::BranchOnTwoConds,
+ {IsAnyExitTaken, IsLatchExitTaken});
// Disconnect the early exit edge.
EarlyExitingVPBB->getTerminator()->eraseFromParent();
More information about the llvm-commits
mailing list