[llvm] Introduce check-first vectorization for early-exit loops. (PR #227201)

Arjun H Kumar via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 28 23:55:42 PDT 2026


https://github.com/arjun-harikumar-amd created https://github.com/llvm/llvm-project/pull/227201

Check-first is a new strategy for vectorizing loops that have
uncountable early exits and also store to memory. The existing
strategies either require the loop to be read-only, or speculate the
whole body and mask its side effects. Check-first instead evaluates the
exit conditions for an entire vector chunk up front and executes the
body only when no lane exits. When an exit fires, the chunk is abandoned
and replayed by the scalar loop, which resumes from the start of that
chunk rather than from the exiting lane.

The vector loop is restructured into a cascade:

-   vector.check       
-   vector.check1..N-1 
-   vector.body        
-   vector.check.exit 

  
 Example:
 IR before:
```
 loop:
    %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
    %x.ptr = getelementptr inbounds [1024 x i32], ptr @x, i64 0, i64 %iv
    %x.val = load i32, ptr %x.ptr, align 4
    %found = icmp ne i32 %x.val, 0
    br i1 %found, label %exit, label %latch

  latch:
    %a.ptr = getelementptr inbounds [1024 x i32], ptr @a, i64 0, i64 %iv
    store i32 1, ptr %a.ptr, align 4
    %iv.next = add nuw nsw i64 %iv, 1
    %done = icmp eq i64 %iv.next, 1024
    br i1 %done, label %exit, label %loop
```

With this patch the vectorized code is emitted as:

  ```
vector.check:               ; preds = %vector.body, %vector.ph
    %index = phi i64 [ 0, %vector.ph ], [ %index.next, %vector.body ]
    %0 = add i64 %index, 1
    %1 = add i64 %index, 2
    %2 = add i64 %index, 3
    %3 = getelementptr inbounds [1024 x i32], ptr @x, i64 0, i64 %index
    %wide.load = load <4 x i32>, ptr %3, align 4
    %4 = icmp ne <4 x i32> %wide.load, zeroinitializer
    %5 = freeze <4 x i1> %4
    %6 = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> %5)
    br i1 %6, label %vector.check.exit, label %vector.body

  vector.body:                ; preds = %vector.check
    %7 = getelementptr inbounds [1024 x i32], ptr @a, i64 0, i64 %index
    %8 = getelementptr inbounds [1024 x i32], ptr @a, i64 0, i64 %0
    %9 = getelementptr inbounds [1024 x i32], ptr @a, i64 0, i64 %1
    %10 = getelementptr inbounds [1024 x i32], ptr @a, i64 0, i64 %2
    store i32 1, ptr %7, align 4
    store i32 1, ptr %8, align 4
    store i32 1, ptr %9, align 4
    store i32 1, ptr %10, align 4
    %index.next = add nuw i64 %index, 4
    %11 = icmp eq i64 %index.next, 1024
    br i1 %11, label %middle.block, label %vector.check

  vector.check.exit:          ; preds = %vector.check
    br label %scalar.ph

  scalar.ph:                  ; preds = %vector.check.exit
    br label %loop

  loop:                       ; preds = %scalar.ph, %latch
    %iv = phi i64 [ %index, %scalar.ph ], [ %iv.next, %latch ]
```

>From 835fecb52cd730c2ddff13ba03bfc5dabc2ffb4a Mon Sep 17 00:00:00 2001
From: Arjun H Kumar <Arjun.HKumar at amd.com>
Date: Tue, 1 Sep 2026 11:54:48 +0530
Subject: [PATCH 1/2] [VPlan] Add infrastructure for check-first early-exit 
 vectorization. Adds the VPlan state and APIs required by check-first
 vectorization of loops with multiple uncountable early exits

---
 llvm/lib/Transforms/Vectorize/VPlan.cpp       | 31 +++++++++
 llvm/lib/Transforms/Vectorize/VPlan.h         | 69 +++++++++++++++++++
 .../Vectorize/VPlanConstruction.cpp           | 17 ++++-
 .../Transforms/Vectorize/VPlanPredicator.cpp  |  2 +
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp | 13 +++-
 .../Transforms/Vectorize/VPlanTransforms.cpp  |  3 +-
 .../Transforms/Vectorize/VPlanVerifier.cpp    |  6 ++
 7 files changed, 137 insertions(+), 4 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlan.cpp b/llvm/lib/Transforms/Vectorize/VPlan.cpp
index 87d24ed9d8d4b..f6addb9254a59 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlan.cpp
@@ -1237,6 +1237,18 @@ static void remapOperands(VPBlockBase *Entry, VPBlockBase *NewEntry,
   }
 }
 
+/// Returns the clone of VPBB Old recorded in Old2NewBlocks, or nullptr when
+/// Old is null.
+static VPBasicBlock *
+remapClonedBlock(const DenseMap<VPBlockBase *, VPBlockBase *> &Old2NewBlocks,
+                 VPBasicBlock *Old) {
+  if (!Old)
+    return nullptr;
+  VPBlockBase *New = Old2NewBlocks.lookup(Old);
+  assert(New && "Check-first block not found in cloned plan.");
+  return cast<VPBasicBlock>(New);
+}
+
 VPlan *VPlan::duplicate() {
   unsigned NumBlocksBeforeCloning = CreatedBlocks.size();
   // Clone blocks.
@@ -1326,6 +1338,25 @@ VPlan *VPlan::duplicate() {
       NewPlan->ExitBlocks.push_back(cast<VPIRBasicBlock>(VPB));
   }
 
+  NewPlan->CheckFirst.UsesMaskedReplay = CheckFirst.UsesMaskedReplay;
+  NewPlan->CheckFirst.InclusiveReplayStores = CheckFirst.InclusiveReplayStores;
+
+  // Map each original block to its clone via a single depth-first traversal.
+  DenseMap<VPBlockBase *, VPBlockBase *> Old2NewBlocks;
+  for (const auto &[OldBB, NewBB] :
+       zip_equal(vp_depth_first_deep(Entry), vp_depth_first_deep(NewEntry)))
+    Old2NewBlocks[OldBB] = NewBB;
+
+  // ExitBlocks must list every exit of the original scalar loop, including
+  // currently unreachable ones, which the traversals above do not reach.
+  for (VPIRBasicBlock *EB : ExitBlocks)
+    if (!Old2NewBlocks.contains(EB))
+      NewPlan->ExitBlocks.push_back(
+          NewPlan->createVPIRBasicBlock(EB->getIRBasicBlock()));
+
+  NewPlan->CheckFirst.ExitBlock =
+      remapClonedBlock(Old2NewBlocks, CheckFirst.ExitBlock);
+
   return NewPlan;
 }
 
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 7d2c2fa1bdd23..4bce8353ee293 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -86,6 +86,10 @@ enum class UncountableExitStyle {
   /// uncountable exit is taken, then all lanes before the exiting lane will
   /// complete, leaving just the final lane to execute in the scalar tail.
   MaskedHandleExitInScalarLoop,
+  /// Check-first semantics: exit conditions are evaluated at the start of each
+  /// vector iteration before any stores execute. On early exit, the scalar loop
+  /// resumes from the start of the current vector chunk.
+  CheckFirst,
 };
 
 /// VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
@@ -4823,6 +4827,27 @@ class VPlan {
   /// VPIRBasicBlock wrapping the header of the original scalar loop.
   VPIRBasicBlock *ScalarHeader;
 
+  /// Used for recording the state for check-first early-exit vectorization.
+  /// Everything else the strategy needs is derived from the plan: the cascade
+  /// header is the vector loop entry, the masked replay recipes live in
+  /// ExitBlock, guarded memory operations carry their guard as the recipe's
+  /// mask operand, and the early exit is the exit block left without
+  /// predecessors while the cascade is wired up.
+  struct CheckFirstEarlyExitState {
+    /// Block routing check-first early exits out of the vector loop. Its
+    /// presence also marks the plan as using the check-first strategy.
+    VPBasicBlock *ExitBlock = nullptr;
+
+    /// Whether the exiting chunk is replayed in vector form under a mask,
+    /// rather than being left to the scalar loop.
+    bool UsesMaskedReplay = false;
+
+    /// Stores that precede the exit condition in the original iteration and so
+    /// must also execute for the exiting lane during masked replay.
+    SmallPtrSet<const Instruction *, 4> InclusiveReplayStores;
+  };
+  CheckFirstEarlyExitState CheckFirst;
+
   /// Immutable list of VPIRBasicBlocks wrapping the exit blocks of the original
   /// scalar loop. Note that some exit blocks may be unreachable at the moment,
   /// e.g. if the scalar epilogue always executes.
@@ -4966,6 +4991,50 @@ class VPlan {
         getScalarHeader()->getSinglePredecessor());
   }
 
+  /// Return the block routing check-first early exits out of the vector loop.
+  /// A non-null result means the plan uses check-first vectorization.
+  VPBasicBlock *getCheckFirstExitBlock() const { return CheckFirst.ExitBlock; }
+
+  void setCheckFirstExitBlock(VPBasicBlock *VPBB) {
+    assert((!CheckFirst.ExitBlock || CheckFirst.ExitBlock == VPBB) &&
+           "CheckFirstExitBlock already set");
+    CheckFirst.ExitBlock = VPBB;
+  }
+
+  /// Return the block holding the masked-replay recipes, which are placed in
+  /// the check-first exit block, or nullptr if the plan replays in the scalar
+  /// loop instead.
+  VPBasicBlock *getCheckFirstMaskedReplayBlock() const {
+    return CheckFirst.UsesMaskedReplay ? CheckFirst.ExitBlock : nullptr;
+  }
+
+  void setCheckFirstUsesMaskedReplay() { CheckFirst.UsesMaskedReplay = true; }
+
+  void addCheckFirstInclusiveReplayStore(const Instruction *I) {
+    CheckFirst.InclusiveReplayStores.insert(I);
+  }
+
+  bool isCheckFirstInclusiveReplayStore(const Instruction *I) const {
+    return CheckFirst.InclusiveReplayStores.contains(I);
+  }
+
+  /// Return the header of the check-first cascade, which is the entry of the
+  /// vector loop. Every cascade block branches to the check-first exit block
+  /// and only the header carries the canonical IV phi, so the header stays
+  /// identifiable once the loop region has been dissolved.
+  VPBasicBlock *getCheckFirstCheckHeaderBlock() const {
+    if (!CheckFirst.ExitBlock)
+      return nullptr;
+    if (const VPRegionBlock *R = getVectorLoopRegion())
+      return cast<VPBasicBlock>(const_cast<VPBlockBase *>(R->getEntry()));
+    for (VPBlockBase *Pred : CheckFirst.ExitBlock->getPredecessors()) {
+      auto *VPBB = cast<VPBasicBlock>(Pred);
+      if (!VPBB->empty() && isa<VPPhi>(&VPBB->front()))
+        return VPBB;
+    }
+    return nullptr;
+  }
+
   /// Return the VPIRBasicBlock wrapping the header of the scalar loop.
   VPIRBasicBlock *getScalarHeader() const { return ScalarHeader; }
 
diff --git a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
index 7f748960b1d8c..bea54fdb0a8f5 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanConstruction.cpp
@@ -462,13 +462,25 @@ static void createLoopRegion(VPlan &Plan, VPBlockBase *HeaderVPB, DebugLoc DL) {
   // Transfer latch's successors to the region.
   VPBlockUtils::transferSuccessors(LatchVPBB, R);
 
+  VPBasicBlock *CheckExitVPBB = Plan.getCheckFirstExitBlock();
+  if (CheckExitVPBB) {
+    assert(CheckExitVPBB->empty() &&
+           "check.exit block should be empty before region creation");
+    assert(CheckExitVPBB->getNumSuccessors() == 0 &&
+           "check.exit should have no successors before temporary edge");
+    VPBlockUtils::connectBlocks(CheckExitVPBB, LatchVPBB);
+  }
+
   VPBlockUtils::connectBlocks(PreheaderVPBB, R);
   R->setEntry(HeaderVPB);
   R->setExiting(LatchVPBB);
 
   // All VPBB's reachable shallowly from HeaderVPB belong to the current region.
-  for (VPBlockBase *VPBB : vp_depth_first_shallow(HeaderVPB))
+  for (VPBlockBase *VPBB : vp_depth_first_shallow(HeaderVPB)) {
+    if (VPBB == Plan.getScalarPreheader())
+      continue;
     VPBB->setParent(R);
+  }
 
   if (!IsOutermost)
     return;
@@ -1302,7 +1314,8 @@ void VPlanTransforms::createLoopRegions(VPlan &Plan, DebugLoc DL) {
 
   VPRegionBlock *TopRegion = Plan.getVectorLoopRegion();
   TopRegion->setName("vector loop");
-  TopRegion->getEntryBasicBlock()->setName("vector.body");
+  TopRegion->getEntryBasicBlock()->setName(
+      Plan.getCheckFirstExitBlock() ? "vector.check" : "vector.body");
 }
 
 void VPlanTransforms::foldTailByMasking(VPlan &Plan) {
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 3f35f14e876f2..59f5745af9745 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -398,6 +398,8 @@ void VPlanTransforms::introduceMasksAndLinearize(VPlan &Plan) {
   // Nested loop regions (outer-loop vectorization) are not supported yet.
   if (Plan.isOuterLoop())
     return;
+  if (Plan.getCheckFirstExitBlock())
+    return;
   VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
   // Scan the body of the loop in a topological order to visit each basic block
   // after having visited its predecessor basic blocks.
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 72c644c1df3ae..63139cec0f0f9 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -96,6 +96,7 @@ bool VPRecipeBase::mayWriteToMemory() const {
   case VPBlendSC:
   case VPReductionEVLSC:
   case VPReductionSC:
+  case VPVectorEndPointerSC:
   case VPVectorPointerSC:
   case VPWidenCanonicalIVSC:
   case VPWidenCastSC:
@@ -151,6 +152,7 @@ bool VPRecipeBase::mayReadFromMemory() const {
   case VPBlendSC:
   case VPReductionEVLSC:
   case VPReductionSC:
+  case VPVectorEndPointerSC:
   case VPVectorPointerSC:
   case VPWidenCanonicalIVSC:
   case VPWidenCastSC:
@@ -875,9 +877,13 @@ Value *VPInstruction::generate(VPTransformState &State) {
         cast<VPBasicBlock>(getParent()->getSuccessors()[1]);
     BasicBlock *SecondIRSucc = State.CFG.VPBB2IRBB.lookup(SecondVPSucc);
     BasicBlock *IRBB = State.CFG.VPBB2IRBB[getParent()];
-    auto *Br = Builder.CreateCondBr(Cond, IRBB, SecondIRSucc);
+    // Placeholder for a successor. Assigned in connectToPredecessors.
+    auto *Br =
+        Builder.CreateCondBr(Cond, IRBB, SecondIRSucc ? SecondIRSucc : IRBB);
     // First successor is always forward, reset it to nullptr.
     Br->setSuccessor(0, nullptr);
+    if (!SecondIRSucc)
+      Br->setSuccessor(1, nullptr);
     IRBB->getTerminator()->eraseFromParent();
     applyMetadata(*Br);
     return Br;
@@ -1572,6 +1578,11 @@ void VPInstruction::addOperand(VPValue *Op) {
            "matching operand 1's type and i1, respectively");
     break;
   }
+  case Instruction::PHI:
+    assert((getNumOperands() == 0 ||
+            Ty == getOperand(0)->getScalarType()) &&
+           "all incoming values must have the same type");
+    break;
   default:
     llvm_unreachable("opcode does not support growing the operand list "
                      "outside of construction");
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index ebd4e91232acb..efbb7a106cf51 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -3203,7 +3203,8 @@ bool VPlanTransforms::handleUncountableEarlyExits(
 
   // Dereferenceability is checked separately for uncountable exit loops with
   // stores, as only the loads contributing to the exit condition need to
-  // be checked.
+  // be checked. ReadOnly needs all loads dereferenceable, whereas CheckFirst
+  // checks only condition-slice loads below.
   if (Style == UncountableExitStyle::ReadOnly &&
       !areAllLoadsDereferenceable(HeaderVPBB, TheLoop, PSE, DT, AC))
     return false;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp b/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
index e36ad81cfae2a..ddaba020baef4 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
@@ -310,6 +310,12 @@ bool VPlanVerifier::verifyVPBasicBlock(const VPBasicBlock *VPBB) {
           continue;
         }
 
+        if (VPBasicBlock *CheckExit =
+                VPBB->getPlan()->getCheckFirstExitBlock()) {
+          if (is_contained(CheckExit->getPredecessors(), VPBB))
+            continue;
+        }
+
         errs() << "Use before def!\n";
 #if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
         VPSlotTracker Tracker(VPBB->getPlan());

>From 4dcb865e4b74c74e9cc1d28666a99dc0731d93b5 Mon Sep 17 00:00:00 2001
From: Arjun H Kumar <Arjun.HKumar at amd.com>
Date: Tue, 29 Sep 2026 12:19:05 +0530
Subject: [PATCH 2/2] [LoopVectorize] Introduce check-first vectorization for
 early-exit loops.

---
 .../Vectorize/LoopVectorizationLegality.h     |  23 +
 .../Vectorize/LoopVectorizationLegality.cpp   | 225 ++++++++-
 .../Vectorize/LoopVectorizationPlanner.h      |   2 +
 .../Transforms/Vectorize/LoopVectorize.cpp    |  94 +++-
 .../Transforms/Vectorize/VPlanLowering.cpp    |   3 +
 .../Transforms/Vectorize/VPlanTransforms.cpp  | 431 ++++++++++++++++++
 .../Transforms/Vectorize/VPlanTransforms.h    |   3 +
 .../check-first-instruction-reorder.ll        | 196 ++++++++
 .../check-first-multi-exit-cascade.ll         | 133 ++++++
 .../LoopVectorize/check-first-single-exit.ll  | 136 ++++++
 .../LoopVectorize/check-first-state-update.ll | 109 +++++
 11 files changed, 1340 insertions(+), 15 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/check-first-instruction-reorder.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/check-first-multi-exit-cascade.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/check-first-single-exit.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/check-first-state-update.ll

diff --git a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
index 7b8b27c6541e1..c032cfd840660 100644
--- a/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
+++ b/llvm/include/llvm/Transforms/Vectorize/LoopVectorizationLegality.h
@@ -434,6 +434,21 @@ class LoopVectorizationLegality {
     return getUncountableExitTrait() == UncountableExitTrait::ReadWrite;
   }
 
+  /// Returns true if every widened exit condition load is
+  /// dereferenceable for the complete trip count.
+  bool exitLoadsAreDereferenceable() const {
+    return AllExitLoadsDereferenceable;
+  }
+
+  /// Returns true if this early exit loop would use the check first strategy if
+  /// enabled. Whether the strategy is then safe to apply is a separate
+  /// question, answered by exitLoadsAreDereferenceable().
+  /// TODO: Read-only loops whose loads are not all dereferenceable could fall
+  /// back to check-first instead of bailing out.
+  bool wouldUseCheckFirstStyle() const {
+    return hasUncountableExitWithSideEffects();
+  }
+
   /// Return true if there is store-load forwarding dependencies.
   bool isSafeForAnyStoreLoadForwardDistances() const {
     return LAI->getDepChecker().isSafeForAnyStoreLoadForwardDistances();
@@ -629,6 +644,10 @@ class LoopVectorizationLegality {
   /// for it.
   bool canUncountableExitConditionLoadBeMoved(BasicBlock *ExitingBlock);
 
+  /// Returns true if the exit conditions can be safely speculated.
+  bool
+  canCheckFirstSpeculateExitConditions(ArrayRef<BasicBlock *> ExitingBlocks);
+
   /// Return true if all of the instructions in the block can be speculatively
   /// executed, and record the loads/stores that require masking.
   /// \p SafePtrs is a list of addresses that are known to be legal and we know
@@ -746,6 +765,10 @@ class LoopVectorizationLegality {
   /// Records whether we have an uncountable early exit in a loop that's
   /// either read-only or read-write.
   UncountableExitTrait UncountableExitType = UncountableExitTrait::None;
+
+  /// Records whether every widened exit condition load is
+  /// dereferenceable for the complete trip count.
+  bool AllExitLoadsDereferenceable = true;
 };
 
 } // namespace llvm
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
index e05147ab7a3e5..32fd9e77ad7d9 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationLegality.cpp
@@ -79,6 +79,12 @@ static cl::opt<bool> EnableHistogramVectorization(
     "enable-histogram-loop-vectorization", cl::init(false), cl::Hidden,
     cl::desc("Enables autovectorization of some loops containing histograms"));
 
+static cl::opt<unsigned> MaxUncountableEarlyExits(
+    "max-uncountable-early-exits", cl::init(4), cl::Hidden,
+    cl::desc(
+        "Maximum number of uncountable early exits a loop may contain to be "
+        "eligible for check-first vectorization."));
+
 /// Maximum vectorization interleave count.
 static const unsigned MaxInterleaveFactor = 16;
 
@@ -1649,6 +1655,35 @@ bool LoopVectorizationLegality::canVectorizeLoopNestCFG(
   return Result;
 }
 
+/// Collects the loads feeding the exit conditions of early-exits.
+/// condition, which check-first widens speculatively.
+static void collectExitConditionSliceLoads(ArrayRef<BasicBlock *> ExitingBlocks,
+                                           Loop *L,
+                                           SmallVectorImpl<LoadInst *> &Out) {
+  SmallPtrSet<Value *, 16> Visited;
+  SmallVector<Value *, 16> Worklist;
+  for (BasicBlock *BB : ExitingBlocks) {
+    auto *Br = dyn_cast<CondBrInst>(BB->getTerminator());
+    assert(Br && "exiting block must terminate with a conditional branch");
+    Worklist.push_back(Br->getCondition());
+  }
+  // Duplicated code. Can we make it reusable?
+  while (!Worklist.empty()) {
+    Value *V = Worklist.pop_back_val();
+    if (!Visited.insert(V).second)
+      continue;
+    auto *I = dyn_cast<Instruction>(V);
+    if (!I || !L->contains(I) || isa<PHINode>(I))
+      continue;
+    if (auto *LI = dyn_cast<LoadInst>(I)) {
+      Out.push_back(LI);
+      continue;
+    }
+    for (Value *Op : I->operands())
+      Worklist.push_back(Op);
+  }
+}
+
 bool LoopVectorizationLegality::isVectorizableEarlyExitLoop() {
   BasicBlock *LatchBB = TheLoop->getLoopLatch();
   if (!LatchBB) {
@@ -1766,10 +1801,15 @@ bool LoopVectorizationLegality::isVectorizableEarlyExitLoop() {
       return false;
     }
   } else {
-    // Check all uncountable exiting blocks for movable loads.
-    for (BasicBlock *ExitingBB : UncountableExitingBlocks) {
-      if (!canUncountableExitConditionLoadBeMoved(ExitingBB))
+    if (EnableCheckFirstVectorization &&
+        !EnableEarlyExitVectorizationWithSideEffects) {
+      if (!canCheckFirstSpeculateExitConditions(UncountableExitingBlocks))
         return false;
+    } else {
+      for (BasicBlock *ExitingBB : UncountableExitingBlocks) {
+        if (!canUncountableExitConditionLoadBeMoved(ExitingBB))
+          return false;
+      }
     }
   }
 
@@ -1787,6 +1827,56 @@ bool LoopVectorizationLegality::isVectorizableEarlyExitLoop() {
     }
   }
 
+  // Safe only if every widened condition slice load is dereferenceable.
+  if (HasSideEffects) {
+    SmallVector<LoadInst *, 4> SpeculatedCondLoads;
+    collectExitConditionSliceLoads(UncountableExitingBlocks, TheLoop,
+                                   SpeculatedCondLoads);
+
+    bool AllDeref = true;
+    for (LoadInst *LI : SpeculatedCondLoads) {
+      if (!isDereferenceableAndAlignedInLoop(LI, TheLoop, *PSE.getSE(), *DT,
+                                             AC)) {
+        AllDeref = false;
+        break;
+      }
+    }
+
+    AllExitLoadsDereferenceable = AllDeref;
+
+    LLVM_DEBUG({
+      dbgs() << "LV: check-first early-exit memory-safety strategy: ";
+      if (AllDeref)
+        dbgs() << "all speculated condition-slice loads provably "
+                  "dereferenceable. \n";
+      else
+        dbgs() << "Condition-slice loads not provably dereferenceable. \n";
+    });
+  }
+
+  bool WillUseCheckFirst =
+      HasSideEffects && !EnableEarlyExitVectorizationWithSideEffects;
+  if (WillUseCheckFirst) {
+    const InductionDescriptor *IndDesc = nullptr;
+    if (Inductions.size() == 1) {
+      IndDesc = &Inductions.begin()->second;
+    } else if (PHINode *PrimaryIV = getPrimaryInduction()) {
+      auto It = Inductions.find(PrimaryIV);
+      if (It != Inductions.end())
+        IndDesc = &It->second;
+    }
+    if (!IndDesc ||
+        (IndDesc->getKind() != InductionDescriptor::IK_IntInduction &&
+         IndDesc->getKind() != InductionDescriptor::IK_PtrInduction) ||
+        !IndDesc->getConstIntStepValue()) {
+      reportVectorizationFailure(
+          "Check-first early-exit vectorization requires a single integer or "
+          "pointer induction with a constant step",
+          "UnsupportedCheckFirstInduction", ORE, TheLoop);
+      return false;
+    }
+  }
+
   [[maybe_unused]] const SCEV *SymbolicMaxBTC =
       PSE.getSymbolicMaxBackedgeTakenCount();
   // Since we have an exact exit count for the latch and the early exit
@@ -1887,6 +1977,135 @@ bool LoopVectorizationLegality::canUncountableExitConditionLoadBeMoved(
   return true;
 }
 
+bool LoopVectorizationLegality::canCheckFirstSpeculateExitConditions(
+    ArrayRef<BasicBlock *> ExitingBlocks) {
+  // Threshold check for the number of early exits.
+  // Should this check be moved to the caller?
+  if (ExitingBlocks.size() > MaxUncountableEarlyExits) {
+    reportVectorizationFailure(
+        "Too many uncountable early exits for check-first vectorization",
+        "TooManyEarlyExitsForCheckFirst", ORE, TheLoop);
+    return false;
+  }
+  // Is this check a duplicate from isVectorizableEarlyExitLoop()? 
+  // BasicBlock *Latch = TheLoop->getLoopLatch();
+  // if (!Latch) {
+  //   reportVectorizationFailure("Early-exit loop has no unique latch",
+  //                              "NoUniqueLatchForCheckFirst", ORE, TheLoop);
+  //   return false;
+  // }
+
+  // Collection of all the load instructions that are part of the early-exit
+  // condition slice.
+  SmallVector<LoadInst *, 8> CondLoads;
+  SmallVector<Value *, 16> Worklist;
+  SmallPtrSet<Value *, 16> Visited;
+  for (BasicBlock *BB : ExitingBlocks) {
+    auto *Br = dyn_cast<CondBrInst>(BB->getTerminator());
+    if (!Br) {
+      reportVectorizationFailure(
+          "Exiting block does not terminate with a conditional branch",
+          "UnsupportedCheckFirstExit", ORE, TheLoop);
+      return false;
+    }
+    Worklist.push_back(Br->getCondition());
+  }
+
+  while (!Worklist.empty()) {
+    Value *V = Worklist.pop_back_val();
+    if (!Visited.insert(V).second)
+      continue;
+    if (TheLoop->isLoopInvariant(V))
+      continue;
+    auto *I = dyn_cast<Instruction>(V);
+    if (!I || !TheLoop->contains(I)) {
+      reportVectorizationFailure(
+          "Early exit condition depends on a value that cannot be "
+          "speculatively evaluated for check-first vectorization",
+          "UnsupportedCheckFirstExitCondition", ORE, TheLoop);
+      return false;
+    }
+    // Conditions for loads.
+    // 1. Load should not be volatile or atomic.
+    // 2. AR should be affine and of the form {start + step}.
+    if (auto *LI = dyn_cast<LoadInst>(I)) {
+      const auto *AR = dyn_cast<SCEVAddRecExpr>(
+          PSE.getSE()->getSCEV(LI->getPointerOperand()));
+      if (!LI->isSimple() || !AR || AR->getLoop() != TheLoop ||
+          !AR->isAffine()) {
+        reportVectorizationFailure(
+            "Early exit condition depends on a load that is not a simple "
+            "affine (unit-stride) access",
+            "CheckFirstExitLoadInvariantAddress", ORE, TheLoop);
+        return false;
+      }
+      CondLoads.push_back(LI);
+      continue;
+    }
+    if (isa<PHINode>(I)) {
+      if (I->getParent() != TheLoop->getHeader()) {
+        reportVectorizationFailure(
+            "Early exit condition depends on a non-header PHI",
+            "UnsupportedCheckFirstExitCondition", ORE, TheLoop);
+        return false;
+      }
+      continue;
+    }
+    if (I->mayReadOrWriteMemory() || !isSafeToSpeculativelyExecute(I)) {
+      reportVectorizationFailure(
+          "Early exit condition contains an operation that cannot be "
+          "speculatively executed",
+          "UnsupportedCheckFirstExitCondition", ORE, TheLoop);
+      return false;
+    }
+    for (Value *Op : I->operands()) {
+      // Skip constants and loop invariants.
+      if (TheLoop->isLoopInvariant(Op))
+        continue;
+      Worklist.push_back(Op);
+    }
+  }
+
+  SmallPtrSet<const Instruction *, 4> CondLoadSet(CondLoads.begin(),
+                                                  CondLoads.end());
+  ConditionallyExecutedOps.clear();
+
+  // Condition loads and instructions that do not touch memory are skipped.
+  // Every other memory operation is recorded in ConditionallyExecutedOps.
+  // isMaskRequired later uses that set so those operations execute only for
+  // lanes that actually reach them. Other loads are allowed. They stay masked
+  // and are not speculated with the exit check. Anything that touches memoryand
+  // is neither a load nor a store is rejected. That includes calls, atomics,
+  // and similar operations. Each store must not alias any condition load.
+  for (auto *BB : TheLoop->blocks()) {
+    for (auto &I : *BB) {
+      if (CondLoadSet.contains(&I) || !I.mayReadOrWriteMemory())
+        continue;
+      ConditionallyExecutedOps.insert(&I);
+      if (isa<LoadInst>(&I))
+        continue;
+      auto *SI = dyn_cast<StoreInst>(&I);
+      if (!SI) {
+        reportVectorizationFailure(
+            "Unsupported memory operation in check-first early-exit loop",
+            "UnsupportedCheckFirstMemOp", ORE, TheLoop);
+        return false;
+      }
+      for (LoadInst *CL : CondLoads) {
+        if (AA->alias(CL->getPointerOperand(), SI->getPointerOperand()) !=
+            AliasResult::NoAlias) {
+          reportVectorizationFailure(
+              "Cannot determine whether an early-exit condition load aliases "
+              "a store (deferred stores must not be observed out of order)",
+              "CheckFirstExitLoadAliasesStore", ORE, TheLoop);
+          return false;
+        }
+      }
+    }
+  }
+  return true;
+}
+
 bool LoopVectorizationLegality::canVectorize(bool UseVPlanNativePath) {
   // Store the result and return it at the end instead of exiting early, in case
   // allowExtraAnalysis is used to report multiple reasons for not vectorizing.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
index 2e5b0f64bfed2..cc04d969cfd60 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.h
@@ -53,6 +53,8 @@ struct VFRange;
 extern cl::opt<bool> EnableVPlanNativePath;
 extern cl::opt<unsigned> ForceTargetInstructionCost;
 extern cl::opt<bool> PreferInLoopReductions;
+extern cl::opt<bool> EnableCheckFirstVectorization;
+extern cl::opt<bool> EnableEarlyExitVectorizationWithSideEffects;
 
 /// \return An upper bound for vscale based on TTI or the vscale_range
 /// attribute.
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 53c9c91e9e377..192f00e57160b 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -406,12 +406,17 @@ static cl::opt<bool> EnableEarlyExitVectorization(
     cl::desc(
         "Enable vectorization of early exit loops with uncountable exits."));
 
-static cl::opt<bool> EnableEarlyExitVectorizationWithSideEffects(
+cl::opt<bool> llvm::EnableEarlyExitVectorizationWithSideEffects(
     "enable-early-exit-vectorization-with-side-effects", cl::init(false),
     cl::Hidden,
     cl::desc("Enable vectorization of early exit loops with uncountable exits "
              "and side effects"));
 
+cl::opt<bool> llvm::EnableCheckFirstVectorization(
+    "enable-check-first-early-exit-vectorization", cl::init(false), cl::Hidden,
+    cl::desc("Enable check-first vectorization of early exit loops with "
+             "multiple exits."));
+
 // Returns true if the epilogue VF has been set to a non-zero value other than
 // VF=1 (scalar).
 static bool hasForcedEpilogueVF() {
@@ -419,6 +424,17 @@ static bool hasForcedEpilogueVF() {
          EpilogueVectorizationForceVF != ElementCount::getFixed(1);
 }
 
+/// Return true when it is legal to use check-first vectorization. Currently,
+/// the fallback mechanism is scalar replay when the early exit is triggered
+/// from the loop.
+static bool usesCheckFirstReplay(const LoopVectorizationLegality *Legal) {
+  if (!Legal->hasUncountableEarlyExit())
+    return false;
+  if (Legal->hasUncountableExitWithSideEffects())
+    return !EnableEarlyExitVectorizationWithSideEffects;
+  return EnableCheckFirstVectorization && Legal->wouldUseCheckFirstStyle();
+}
+
 // Likelyhood of bypassing the vectorized loop because there are zero trips left
 // after prolog. See `emitIterationCountCheck`.
 static constexpr uint32_t MinItersBypassWeights[] = {1, 127};
@@ -3595,6 +3611,11 @@ LoopVectorizationPlanner::selectInterleaveCount(VPlan &Plan, ElementCount VF,
   if (Plan.hasEarlyExit())
     return 1;
 
+  // Interleaving would break check-first scalar-replay resume wiring.
+  // So forcing IC=1.
+  if (usesCheckFirstReplay(Legal))
+    return 1;
+
   const bool HasReductions =
       any_of(Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis(),
              IsaPred<VPReductionPHIRecipe>);
@@ -5413,6 +5434,10 @@ void LoopVectorizationPlanner::plan(ElementCount UserVF, unsigned UserIC) {
   if (!MaxFactors) // Cases that should not to be vectorized nor interleaved.
     return;
 
+  // Disable scalable vectorization for check-first early-exit loops for now.
+  if (usesCheckFirstReplay(Legal))
+    MaxFactors.ScalableVF = ElementCount::getScalable(0);
+
   Config.collectInLoopReductions();
   // Cases that may be vectorized may be optimized by unit stride predicates.
   // TODO: Currently unit stride predicates are added unconditionally, even if
@@ -5934,6 +5959,9 @@ DenseMap<const SCEV *, Value *> LoopVectorizationPlanner::executePlan(
   // removes unneeded loop regions first.
   const bool HasTailFolded = BestVPlan.hasTailFolded();
   RUN_VPLAN_PASS(VPlanTransforms::dissolveLoopRegions, BestVPlan);
+  // Scalar replay routes check.exit to the scalar preheader after region
+  // dissolution.
+  VPlanTransforms::wireCheckFirstExitToScalar(BestVPlan);
   // Expand BranchOnTwoConds after dissolution, when latch has direct access to
   // its successors.
   RUN_VPLAN_PASS(VPlanTransforms::expandBranchOnTwoConds, BestVPlan);
@@ -6539,10 +6567,19 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
   //       the loop inside handleUncountableEarlyExits itself.
   if (Legal->hasUncountableEarlyExit()) {
     // TODO: Check target preference for style.
-    UncountableExitStyle EEStyle =
-        Legal->hasUncountableExitWithSideEffects()
-            ? UncountableExitStyle::MaskedHandleExitInScalarLoop
-            : UncountableExitStyle::ReadOnly;
+    UncountableExitStyle EEStyle;
+    if (!Legal->hasUncountableExitWithSideEffects())
+      EEStyle = UncountableExitStyle::ReadOnly;
+    else if (EnableEarlyExitVectorizationWithSideEffects)
+      EEStyle = UncountableExitStyle::MaskedHandleExitInScalarLoop;
+    else
+      EEStyle = UncountableExitStyle::CheckFirst;
+
+    assert((EEStyle != UncountableExitStyle::CheckFirst ||
+            Legal->exitLoadsAreDereferenceable()) &&
+           "check-first vectorization reached for a loop whose speculatively "
+           "widened loads could not be made memory-safe");
+
     if (!RUN_VPLAN_PASS(VPlanTransforms::handleUncountableEarlyExits, *VPlan0,
                         OrigLoop, PSE, *DT, Legal->getAssumptionCache(),
                         EEStyle))
@@ -6553,7 +6590,9 @@ VPlanPtr LoopVectorizationPlanner::tryToBuildVPlan1() {
 
   RUN_VPLAN_PASS(VPlanTransforms::createLoopRegions, *VPlan0,
                  getDebugLocFromInstOrOperands(Legal->getPrimaryInduction()));
-  if (CM->foldTailByMasking())
+  // Check-first plans manage their own exit masking; tail folding would
+  // interfere with the cascade's resume wiring.
+  if (CM->foldTailByMasking() && !VPlan0->getCheckFirstExitBlock())
     RUN_VPLAN_PASS(VPlanTransforms::foldTailByMasking, *VPlan0);
   RUN_VPLAN_PASS(VPlanTransforms::introduceMasksAndLinearize, *VPlan0);
 
@@ -7270,8 +7309,12 @@ static void checkMixedPrecision(Loop *L, OptimizationRemarkEmitter *ORE) {
 /// TODO: This is currently overly pessimistic because the loop may not take
 /// the early exit, but better to keep this conservative for now. In future,
 /// it might be possible to relax this by using branch probabilities.
+///
+/// For check first loops, add the scalar replay cost of the early exit chunk.
 static InstructionCost calculateEarlyExitCost(VPCostContext &CostCtx,
-                                              VPlan &Plan, ElementCount VF) {
+                                              VPlan &Plan, ElementCount VF,
+                                              uint64_t ScalarCostPerIter,
+                                              bool IsCheckFirstReplay) {
   InstructionCost Cost = 0;
   for (auto *ExitVPBB : Plan.getExitBlocks()) {
     for (auto *PredVPBB : ExitVPBB->getPredecessors()) {
@@ -7285,6 +7328,16 @@ static InstructionCost calculateEarlyExitCost(VPCostContext &CostCtx,
       }
     }
   }
+
+  if (IsCheckFirstReplay && VF.isFixed()) {
+    // At most VF iterations are replayed.
+    uint64_t ReplayedIters = VF.getFixedValue();
+    InstructionCost ReplayCost(ScalarCostPerIter * ReplayedIters);
+    LLVM_DEBUG(dbgs() << "LV: Adding check-first scalar-replay cost "
+                      << ReplayCost << " (~" << ReplayedIters
+                      << " scalar iterations) for VF " << VF << ".\n");
+    Cost += ReplayCost;
+  }
   return Cost;
 }
 
@@ -7301,7 +7354,8 @@ static bool isOutsideLoopWorkProfitable(GeneratedRTChecks &Checks,
                                         PredicatedScalarEvolution &PSE,
                                         VPCostContext &CostCtx, VPlan &Plan,
                                         EpilogueLowering SEL,
-                                        std::optional<unsigned> VScale) {
+                                        std::optional<unsigned> VScale,
+                                        bool IsCheckFirstReplay) {
   InstructionCost RtC = Checks.getCost();
   if (!RtC.isValid())
     return false;
@@ -7327,8 +7381,9 @@ static bool isOutsideLoopWorkProfitable(GeneratedRTChecks &Checks,
 
   InstructionCost TotalCost = RtC;
   // Add on the cost of any work required in the vector early exit block, if
-  // one exists.
-  TotalCost += calculateEarlyExitCost(CostCtx, Plan, VF.Width);
+  // one exists plus check first scalar replay cost.
+  TotalCost += calculateEarlyExitCost(CostCtx, Plan, VF.Width, ScalarC,
+                                      IsCheckFirstReplay);
   TotalCost += Plan.getMiddleBlock()->cost(VF.Width, CostCtx);
 
   // First, compute the minimum iteration count required so that the vector
@@ -7910,6 +7965,19 @@ bool LoopVectorizePass::processLoop(Loop *L) {
     return false;
   }
 
+  // Memory-safety gate: bail to the scalar loop when a speculatively widened
+  // exit-condition load is not provably dereferenceable.
+  if (EnableCheckFirstVectorization &&
+      !EnableEarlyExitVectorizationWithSideEffects &&
+      LVL.wouldUseCheckFirstStyle() && !LVL.exitLoadsAreDereferenceable()) {
+    reportVectorizationFailure(
+        "check-first early-exit memory-safety strategy is Unsafe: a "
+        "speculatively-widened condition load could not be proven "
+        "dereferenceable for the full trip count",
+        "CheckFirstUnsafeMemSafety", ORE, L);
+    return false;
+  }
+
   bool IsInnerLoop = L->isInnermost();
 
   // Outer loops require a computable trip count.
@@ -7926,7 +7994,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
       return false;
     }
     if (LVL.hasUncountableExitWithSideEffects() &&
-        !EnableEarlyExitVectorizationWithSideEffects) {
+        !EnableEarlyExitVectorizationWithSideEffects &&
+        !EnableCheckFirstVectorization) {
       reportVectorizationFailure("Auto-vectorization of loops with uncountable "
                                  "early exit and side effects is not enabled",
                                  "UncountableEarlyExitSideEffectLoopsDisabled",
@@ -8110,7 +8179,8 @@ bool LoopVectorizePass::processLoop(Loop *L) {
                           /*ReusePrintingSlotTracker=*/true);
     if (!ForceVectorization &&
         !isOutsideLoopWorkProfitable(Checks, VF, L, PSE, CostCtx, *BestPlanPtr,
-                                     SEL, Config.getVScaleForTuning())) {
+                                     SEL, Config.getVScaleForTuning(),
+                                     usesCheckFirstReplay(&LVL))) {
       ORE->emit([&]() {
         return OptimizationRemarkAnalysisAliasing(
                    DEBUG_TYPE, "CantReorderMemOps", L->getStartLoc(),
diff --git a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
index e386fffe0880f..8e670e68f8032 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
@@ -767,6 +767,9 @@ void VPlanTransforms::materializeConstantVectorTripCount(
       !isa<VPIRValue>(TC))
     return;
 
+  if (Plan.getCheckFirstExitBlock())
+    return;
+
   // Materialize vector trip counts for constants early if it can simply
   // be computed as (Original TC / VF * UF) * VF * UF.
   // TODO: Compute vector trip counts for loops requiring a scalar epilogue and
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index efbb7a106cf51..a40fa5887db9e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -32,10 +32,12 @@
 #include "llvm/Analysis/MemoryLocation.h"
 #include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
 #include "llvm/Analysis/ScopedNoAliasAA.h"
+#include "llvm/Analysis/ValueTracking.h"
 #include "llvm/Analysis/VectorUtils.h"
 #include "llvm/IR/Intrinsics.h"
 #include "llvm/IR/Metadata.h"
 #include "llvm/Support/Casting.h"
+#include "llvm/Support/ErrorHandling.h"
 #include "llvm/Support/TypeSize.h"
 #include "llvm/Transforms/Utils/LoopUtils.h"
 
@@ -2276,6 +2278,15 @@ static bool cannotHoistOrSinkRecipe(VPRecipeBase &R, VPBasicBlock *FirstBB,
       match(&R, m_Intrinsic<Intrinsic::assume>()))
     return vputils::cannotHoistOrSinkRecipe(R, Sinking);
 
+  bool InSingleSuccChain = false;
+  for (VPBlockBase *Succ = FirstBB; Succ; Succ = Succ->getSingleSuccessor())
+    if (Succ == LastBB) {
+      InSingleSuccChain = true;
+      break;
+    }
+  if (!InSingleSuccChain)
+    return true;
+
   // Check that the memory operation doesn't alias between FirstBB and LastBB.
   auto MemLoc = vputils::getMemoryLocation(R);
 
@@ -3191,6 +3202,29 @@ static bool handleUncountableExitsWithSideEffects(
   return true;
 }
 
+/// Walk backward from ExitCond to collect the recipes needed to evaluate the
+/// exit condition, stopping at PHIs. Returns false if the condition depends on
+/// a memory-writing recipe which cannot be placed in the check block.
+static bool computeConditionSlice(VPValue *ExitCond,
+                                  SmallPtrSetImpl<VPRecipeBase *> &Slice) {
+  SmallVector<VPValue *, 16> Worklist;
+  Worklist.push_back(ExitCond);
+  while (!Worklist.empty()) {
+    VPValue *V = Worklist.pop_back_val();
+    VPRecipeBase *DefR = V->getDefiningRecipe();
+    if (!DefR || DefR->isPhi())
+      continue;
+    if (Slice.contains(DefR))
+      continue;
+    if (DefR->mayWriteToMemory())
+      return false;
+    Slice.insert(DefR);
+    for (VPValue *Op : DefR->operands())
+      Worklist.push_back(Op);
+  }
+  return true;
+}
+
 bool VPlanTransforms::handleUncountableEarlyExits(
     VPlan &Plan, Loop *TheLoop, PredicatedScalarEvolution &PSE,
     DominatorTree &DT, AssumptionCache *AC, UncountableExitStyle Style) {
@@ -3265,6 +3299,243 @@ bool VPlanTransforms::handleUncountableEarlyExits(
              "RPO sort must place dominating exits before dominated ones");
 #endif
 
+  if (Style == UncountableExitStyle::CheckFirst) {
+    // Check-first evaluates the early-exit conditions for the whole vector
+    // iteration before any side effect of that iteration takes place. For
+    //
+    //   for (i = 0; i < 1024; ++i) {
+    //     a[i] = 0;
+    //     if (x[i])
+    //       break;
+    //   }
+    //
+    // the incoming plan interleaves the store with the exit condition:
+    //
+    // loop:
+    //   ir<%iv> = WIDEN-INDUCTION nuw nsw ir<0>, ir<1>, vp<%0>
+    //   EMIT ir<%a.ptr> = getelementptr inbounds ir<@a>, ir<0>, ir<%iv>
+    //   EMIT store ir<0>, ir<%a.ptr>
+    //   EMIT ir<%x.ptr> = getelementptr inbounds ir<@x>, ir<0>, ir<%iv>
+    //   EMIT-SCALAR ir<%x.value> = load ir<%x.ptr>
+    //   EMIT ir<%exit.early> = icmp ne ir<%x.value>, ir<0>
+    //   EMIT branch-on-cond ir<%exit.early>
+    // Successor(s): ir-bb<exit>, latch
+    //
+    // latch:
+    //   EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+    //   EMIT ir<%done> = icmp eq ir<%iv.next>, ir<1024>
+    //   EMIT branch-on-cond ir<%done>
+    // Successor(s): middle.block, loop
+    //
+    // The recipes computing %exit.early stay in the header, which becomes the
+    // first check block and branches to vector.check.exit (resuming the scalar
+    // loop) when any lane exits. Everything else, including the latch recipes,
+    // sinks into a new vector.body that only runs if no lane exits:
+    //
+    // loop:
+    //   ir<%iv> = WIDEN-INDUCTION nuw nsw ir<0>, ir<1>, vp<%0>
+    //   EMIT ir<%x.ptr> = getelementptr inbounds ir<@x>, ir<0>, ir<%iv>
+    //   EMIT-SCALAR ir<%x.value> = load ir<%x.ptr>
+    //   EMIT ir<%exit.early> = icmp ne ir<%x.value>, ir<0>
+    //   EMIT vp<%2> = masked-cond ir<%exit.early>
+    //   EMIT vp<%3> = any-of vp<%2>
+    //   EMIT branch-on-cond vp<%3>
+    // Successor(s): vector.check.exit, vector.body
+    //
+    // vector.body:
+    //   EMIT ir<%a.ptr> = getelementptr inbounds ir<@a>, ir<0>, ir<%iv>
+    //   EMIT store ir<0>, ir<%a.ptr>
+    //   EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+    //   EMIT ir<%done> = icmp eq ir<%iv.next>, ir<1024>
+    //   EMIT branch-on-cond ir<%done>
+    // Successor(s): middle.block, loop
+    //
+    // With N early exits the header is followed by N-1 further vector.check
+    // blocks, each holding one exit's condition slice and branching to
+    // vector.check.exit or on to the next check.
+
+    // Keeping it simple for now. Only support the IV as resume value from the loop.
+    if (range_size(TheLoop->getHeader()->phis()) > 1)
+      return false;
+
+    SmallPtrSet<const VPBlockBase *, 4> ExitBlockSet;
+    SmallPtrSet<const VPBlockBase *, 4> EarlyExitingSet;
+    for (const EarlyExitInfo &E : Exits) {
+      ExitBlockSet.insert(E.EarlyExitVPBB);
+      EarlyExitingSet.insert(E.EarlyExitingVPBB);
+    }
+    // Latch should be a countable exit.
+    if (EarlyExitingSet.contains(LatchVPBB))
+      return false;
+
+    // Returns the single successor of VPBB that stays inside the loop, or
+    // nullptr if there is not exactly one. Edges to the middle block or to an
+    // early-exit block leave the loop and are not part of the chain.
+    auto GetUniqueInLoopSuccessor = [&](VPBasicBlock *VPBB) -> VPBasicBlock * {
+      VPBasicBlock *InLoopSucc = nullptr;
+      for (VPBlockBase *Succ : VPBB->getSuccessors()) {
+        if (Succ == MiddleVPBB || ExitBlockSet.contains(Succ))
+          continue;
+        auto *SuccVPBB = dyn_cast<VPBasicBlock>(Succ);
+        if (!SuccVPBB || InLoopSucc)
+          return nullptr;
+        InLoopSucc = SuccVPBB;
+      }
+      return InLoopSucc;
+    };
+
+    // Collect the chain of blocks from the header to the latch.
+    // The recipes of these blocks are redistributed into the check blocks and
+    // the body block, so anything other than a straight-line chain is
+    // unsupported.
+    SmallVector<VPBasicBlock *, 4> LoopChain;
+    SmallPtrSet<VPBasicBlock *, 8> VisitedBlocks;
+    VPBasicBlock *Cur = HeaderVPBB;
+    while (Cur) {
+      // Revisiting a block means the chain contains a nested cycle.
+      if (!VisitedBlocks.insert(Cur).second)
+        return false;
+
+      // Only the header may hold phis. the other blocks get flattened into it.
+      if (Cur != HeaderVPBB && !Cur->empty() && Cur->begin()->isPhi())
+        return false;
+
+      LoopChain.push_back(Cur);
+      if (Cur == LatchVPBB)
+        break;
+
+      Cur = GetUniqueInLoopSuccessor(Cur);
+      if (!Cur)
+        return false;
+    }
+
+    ArrayRef<VPBasicBlock *> Intermediates =
+        ArrayRef(LoopChain).drop_front().drop_back();
+
+    // All early exits must be inside the loop chain.
+    for (const EarlyExitInfo &Exit : Exits)
+      if (!is_contained(LoopChain, Exit.EarlyExitingVPBB))
+        return false;
+
+    // Compute each exit's condition slice.
+    SmallVector<SmallPtrSet<VPRecipeBase *, 16>, 4> Slices(Exits.size());
+    DenseMap<VPRecipeBase *, unsigned> EarliestCheck;
+    for (unsigned K = 0, E = Exits.size(); K != E; ++K) {
+      if (!computeConditionSlice(Exits[K].CondToExit, Slices[K]))
+        return false;
+      for (VPRecipeBase *R : Slices[K])
+        EarliestCheck.try_emplace(R, K);
+    }
+
+    // From here, all modifications are destructive. We cannot bail out.
+
+    // Detach each original early exit. The checks below take over branching to
+    // the scalar loop. The recipes computing the exit conditions stay in place,
+    // as the check cascade still needs them.
+    for (const EarlyExitInfo &Exit : Exits) {
+      for (VPRecipeBase &R : Exit.EarlyExitVPBB->phis())
+        cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
+      // Erase the branch-on-cond terminating the exiting block.
+      Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
+      // Drop the corresponding CFG edge, e.g. loop -> ir-bb<exit>.
+      VPBlockUtils::disconnectBlocks(Exit.EarlyExitingVPBB, Exit.EarlyExitVPBB);
+    }
+
+    // Flatten intermediate blocks recipes into the header, then connect the
+    // header straight to the latch. Single-exit chains are left untouched.
+    if (!Intermediates.empty()) {
+      // Erase the terminator of the header if it is not an early exit.
+      if (!EarlyExitingSet.contains(HeaderVPBB))
+        HeaderVPBB->getTerminator()->eraseFromParent();
+      // Erase the terminators of the intermediate blocks if they are not early exits.
+      for (VPBasicBlock *BB : Intermediates)
+        if (!EarlyExitingSet.contains(BB) && BB->getTerminator())
+          BB->getTerminator()->eraseFromParent();
+      // Move the recipes of the intermediate blocks to the header in chain
+      // order, so def use is preserved.
+      for (VPBasicBlock *BB : Intermediates)
+        for (VPRecipeBase &R : make_early_inc_range(*BB))
+          R.moveBefore(*HeaderVPBB, HeaderVPBB->end());
+      for (VPBlockBase *S : to_vector(HeaderVPBB->getSuccessors()))
+        VPBlockUtils::disconnectBlocks(HeaderVPBB, S);
+      for (VPBasicBlock *BB : Intermediates)
+        for (VPBlockBase *S : to_vector(BB->getSuccessors()))
+          VPBlockUtils::disconnectBlocks(BB, S);
+      VPBlockUtils::connectBlocks(HeaderVPBB, LatchVPBB);
+    }
+
+    // Create the check cascade. Checks[0] is the header.
+    // Checks[k] is a fresh block holding exit k's condition slice.
+    SmallVector<VPBasicBlock *, 4> Checks;
+    Checks.push_back(HeaderVPBB);
+    for (unsigned K = 1, E = Exits.size(); K != E; ++K)
+      Checks.push_back(Plan.createVPBasicBlock("vector.check"));
+
+    // Body block holds non-slice recipes. Runs when no exit fires.
+    VPBasicBlock *BodyVPBB = Plan.createVPBasicBlock("vector.body");
+
+    // Partition recipes. Slice recipes to their check block, everything else to
+    // the body.
+    auto PartitionBlock = [&](VPBasicBlock *BB) {
+      for (VPRecipeBase &R : make_early_inc_range(*BB)) {
+        if (R.isPhi() || &R == BB->getTerminator())
+          continue;
+        auto It = EarliestCheck.find(&R);
+        VPBasicBlock *Target =
+            It != EarliestCheck.end() ? Checks[It->second] : BodyVPBB;
+        if (Target != BB)
+          R.moveBefore(*Target, Target->end());
+      }
+    };
+    PartitionBlock(HeaderVPBB);
+    if (HeaderVPBB != LatchVPBB)
+      PartitionBlock(LatchVPBB);
+
+    // Routes to the scalar preheader.
+    VPBasicBlock *EarlyExitToScalarVPBB =
+        Plan.createVPBasicBlock("vector.check.exit");
+    Plan.setCheckFirstExitBlock(EarlyExitToScalarVPBB);
+
+    // Extract the latch condition before erasing the latch terminator.
+    auto *LatchBranch = cast<VPInstruction>(LatchVPBB->getTerminator());
+    assert(LatchBranch->getOpcode() == VPInstruction::BranchOnCond &&
+           "Unexpected terminator");
+    VPValue *IsLatchExitTaken = LatchBranch->getOperand(0);
+    DebugLoc LatchDL = LatchBranch->getDebugLoc();
+    LatchBranch->eraseFromParent();
+
+    if (HeaderVPBB != LatchVPBB) {
+      for (VPBlockBase *Succ : to_vector(LatchVPBB->getSuccessors()))
+        VPBlockUtils::disconnectBlocks(LatchVPBB, Succ);
+    }
+
+    for (VPBlockBase *Succ : to_vector(HeaderVPBB->getSuccessors()))
+      VPBlockUtils::disconnectBlocks(HeaderVPBB, Succ);
+
+    // Wire each check.k to exit if exit k fires, else fall through to the next
+    // check or body. BranchOnCond takes successor 0 when true: wire exit first.
+    for (unsigned K = 0, E = Exits.size(); K != E; ++K) {
+      VPBasicBlock *CheckBB = Checks[K];
+      VPBuilder CheckBuilder(CheckBB, CheckBB->end());
+      VPValue *IsExitTaken = CheckBuilder.createNaryOp(VPInstruction::AnyOf,
+                                                       {Exits[K].CondToExit});
+      CheckBuilder.createNaryOp(VPInstruction::BranchOnCond, {IsExitTaken});
+      VPBasicBlock *NextBB = (K + 1 != E) ? Checks[K + 1] : BodyVPBB;
+      VPBlockUtils::connectBlocks(CheckBB, EarlyExitToScalarVPBB);
+      VPBlockUtils::connectBlocks(CheckBB, NextBB);
+    }
+
+    VPBuilder BodyBuilder(BodyVPBB, BodyVPBB->end());
+    BodyBuilder.createNaryOp(VPInstruction::BranchOnCond, {IsLatchExitTaken},
+                             LatchDL);
+
+    // Wire: BodyVPBB → {MiddleVPBB , HeaderVPBB}
+    VPBlockUtils::connectBlocks(BodyVPBB, MiddleVPBB);
+    VPBlockUtils::connectBlocks(BodyVPBB, HeaderVPBB);
+
+    return true;
+  }
+
   // Build the AnyOf condition for the latch terminator using logical OR
   // to avoid poison propagation from later exit conditions when an earlier
   // exit is taken.
@@ -3427,6 +3698,166 @@ bool VPlanTransforms::handleUncountableEarlyExits(
   return true;
 }
 
+/// Returns true if Root transitively uses Target through its defining
+/// recipes operands.
+static bool vpValueDependsOn(VPValue *Root, VPValue *Target) {
+  SmallVector<VPValue *, 16> Worklist{Root};
+  SmallPtrSet<VPValue *, 16> Visited;
+  while (!Worklist.empty()) {
+    VPValue *V = Worklist.pop_back_val();
+    if (V == Target)
+      return true;
+    if (!Visited.insert(V).second)
+      continue;
+    if (VPRecipeBase *Def = V->getDefiningRecipe())
+      for (VPValue *Op : Def->operands())
+        Worklist.push_back(Op);
+  }
+  return false;
+}
+
+static VPValue *
+rebuildIVResumeExprImpl(VPBuilder &B, VPValue *V, VPValue *VectorTC,
+                        VPValue *NewIndex,
+                        SmallDenseMap<VPValue *, VPValue *> &Cache) {
+  using namespace VPlanPatternMatch;
+  if (V == VectorTC)
+    return NewIndex;
+  if (auto It = Cache.find(V); It != Cache.end())
+    return It->second;
+  auto Remap = [&](VPValue *Op) {
+    return rebuildIVResumeExprImpl(B, Op, VectorTC, NewIndex, Cache);
+  };
+  auto RemapBinOp = [&](unsigned Opcode, VPValue *LHS, VPValue *RHS,
+                        DebugLoc DL, const Twine &Name) -> VPValue * {
+    VPValue *NL = Remap(LHS), *NR = Remap(RHS);
+    if (NL == LHS && NR == RHS)
+      return V;
+    auto Flags =
+        cast<VPRecipeWithIRFlags>(V->getDefiningRecipe())->getNoWrapFlags();
+    return B.createOverflowingOp(Opcode, {NL, NR}, Flags, DL, Name);
+  };
+  auto RemapCast = [&](Instruction::CastOps Opcode, VPValue *Op) -> VPValue * {
+    VPValue *NO = Remap(Op);
+    if (NO == Op)
+      return V;
+    return B.createScalarCast(Opcode, NO, V->getScalarType(), DebugLoc());
+  };
+
+  VPValue *A, *Bv;
+  VPValue *Result = V;
+  if (match(V, m_VPInstruction<VPInstruction::PtrAdd>(m_VPValue(A),
+                                                      m_VPValue(Bv)))) {
+    VPValue *NB = Remap(Bv);
+    if (NB != Bv)
+      Result = B.createPtrAdd(A, NB, DebugLoc(), "check.exit.iv.resume");
+  } else if (match(V, m_Mul(m_VPValue(A), m_VPValue(Bv)))) {
+    Result = RemapBinOp(Instruction::Mul, A, Bv, DebugLoc::getUnknown(), "");
+  } else if (match(V, m_c_Add(m_VPValue(A), m_VPValue(Bv)))) {
+    Result =
+        RemapBinOp(Instruction::Add, A, Bv, DebugLoc(), "check.exit.iv.resume");
+  } else if (match(V, m_Sub(m_VPValue(A), m_VPValue(Bv)))) {
+    Result = RemapBinOp(Instruction::Sub, A, Bv, DebugLoc::getUnknown(), "");
+  } else if (match(V, m_Trunc(m_VPValue(A)))) {
+    Result = RemapCast(Instruction::Trunc, A);
+  } else if (match(V, m_ZExt(m_VPValue(A)))) {
+    Result = RemapCast(Instruction::ZExt, A);
+  } else if (match(V, m_SExt(m_VPValue(A)))) {
+    Result = RemapCast(Instruction::SExt, A);
+  } else if (vpValueDependsOn(V, VectorTC)) {
+    assert(false && "Unhandled VectorTC-dependent check-first resume "
+                    "value");
+  }
+  Cache[V] = Result;
+  return Result;
+}
+
+/// Returns the induction resume expression Expr rebuilt with VectorTC
+/// replaced by NewIndex.
+static VPValue *rebuildIVResumeExpr(VPBuilder &B, VPValue *Expr,
+                                    VPValue *VectorTC, VPValue *NewIndex) {
+  SmallDenseMap<VPValue *, VPValue *> Cache;
+  return rebuildIVResumeExprImpl(B, Expr, VectorTC, NewIndex, Cache);
+}
+
+void VPlanTransforms::wireCheckFirstExitToScalar(VPlan &Plan) {
+  VPBasicBlock *CheckExitVPBB = Plan.getCheckFirstExitBlock();
+  if (!CheckExitVPBB)
+    return;
+
+  VPBasicBlock *HeaderVPBB = Plan.getCheckFirstCheckHeaderBlock();
+  assert(HeaderVPBB && "check-first cascade header block not recorded");
+
+  // Replace the temporary check.exit→body edge with check.exit→ScalarPH.
+  assert(CheckExitVPBB->getNumSuccessors() == 1 &&
+         "check.exit should have exactly one successor after dissolution");
+  VPBlockBase *OldSucc = CheckExitVPBB->getSuccessors()[0];
+  VPBlockUtils::disconnectBlocks(CheckExitVPBB, OldSucc);
+
+  VPBasicBlock *ScalarPH = Plan.getScalarPreheader();
+  assert(ScalarPH &&
+         "CheckFirst requires a scalar preheader for early-exit replay. "
+         "Ensure the scalar tail is not removed by earlier passes.");
+  VPBlockUtils::connectBlocks(CheckExitVPBB, ScalarPH);
+
+  assert(HeaderVPBB->getNumPredecessors() == 2 &&
+         "loop header must have exactly two predecessors (preheader, latch)");
+  VPBasicBlock *LatchVPBB = nullptr;
+  for (VPBlockBase *Pred : HeaderVPBB->getPredecessors()) {
+    auto *PredVPBB = cast<VPBasicBlock>(Pred);
+    if (any_of(PredVPBB->getSuccessors(),
+               [&](VPBlockBase *S) { return S != HeaderVPBB; })) {
+      LatchVPBB = PredVPBB;
+      break;
+    }
+  }
+  assert(LatchVPBB &&
+         "could not identify the loop latch (backedge source) among the "
+         "header's predecessors");
+  VPValue *CanonIV = cast<VPPhi>(&*HeaderVPBB->begin());
+
+  auto SPHPhis = ScalarPH->phis();
+  assert(range_size(SPHPhis) == 1 &&
+         "CheckFirst expects exactly one scalar-preheader PHI (the IV). "
+         "Extending to multiple inductions or live-outs requires computing "
+         "proper resume values for each PHI.");
+
+  VPBasicBlock *MiddleVPBB = nullptr;
+  for (VPBlockBase *Succ : LatchVPBB->getSuccessors()) {
+    if (Succ != HeaderVPBB) {
+      MiddleVPBB = cast<VPBasicBlock>(Succ);
+      break;
+    }
+  }
+
+  VPBuilder CheckExitBuilder(CheckExitVPBB, CheckExitVPBB->getFirstNonPhi());
+  VPValue *VectorTC = &Plan.getVectorTripCount();
+
+  using namespace VPlanPatternMatch;
+  for (VPRecipeBase &R : SPHPhis) {
+    auto *Phi = cast<VPPhi>(&R);
+
+    VPValue *MidVal = nullptr;
+    for (unsigned I = 0, E = Phi->getNumIncoming(); I != E; ++I) {
+      if (MiddleVPBB && Phi->getIncomingBlock(I) == MiddleVPBB) {
+        MidVal = Phi->getIncomingValue(I);
+        break;
+      }
+    }
+
+    VPValue *ResumeVal = MidVal ? rebuildIVResumeExpr(CheckExitBuilder, MidVal,
+                                                      VectorTC, CanonIV)
+                                : CanonIV;
+
+    assert(!(MidVal && ResumeVal == MidVal &&
+             vpValueDependsOn(MidVal, VectorTC)) &&
+           "check-first early-exit resume value could not be rebuilt from the "
+           "vector trip count");
+
+    Phi->addIncoming(ResumeVal);
+  }
+}
+
 /// This function tries convert extended in-loop reductions to
 /// VPExpressionRecipe and clamp the \p Range if it is beneficial and
 /// valid. The created recipe must be decomposed to its constituent
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
index 006f3a2d1aa1b..9854b299e164c 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.h
@@ -388,6 +388,9 @@ struct VPlanTransforms {
   /// Disconnect countable early exits from the loop.
   LLVM_ABI_FOR_TEST static void handleCountableEarlyExits(VPlan &Plan);
 
+  /// Connects check-first early-exit blocks to the scalar preheader.
+  static void wireCheckFirstExitToScalar(VPlan &Plan);
+
   /// Replaces the exit condition from
   ///   (branch-on-cond eq CanonicalIVInc, VectorTripCount)
   /// to
diff --git a/llvm/test/Transforms/LoopVectorize/check-first-instruction-reorder.ll b/llvm/test/Transforms/LoopVectorize/check-first-instruction-reorder.ll
new file mode 100644
index 0000000000000..64bfe82bd8844
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/check-first-instruction-reorder.ll
@@ -0,0 +1,196 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 -enable-check-first-early-exit-vectorization | FileCheck %s
+
+;   for (i = 0; i < 1024; i++) {
+;     A[i] = B[i] + C[i];
+;     if (X[i])
+;       break;
+;     D[i] = E[i] + F[i];
+;   }
+;
+ at A = global [1024 x i32] zeroinitializer
+ at B = global [1024 x i32] zeroinitializer
+ at C = global [1024 x i32] zeroinitializer
+ at D = global [1024 x i32] zeroinitializer
+ at E = global [1024 x i32] zeroinitializer
+ at F = global [1024 x i32] zeroinitializer
+ at X = global [1024 x i32] zeroinitializer
+
+define void @instruction_reorder() {
+; CHECK-LABEL: define void @instruction_reorder() {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    br label %[[VECTOR_CHECK:.*]]
+; CHECK:       [[VECTOR_CHECK]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY:.*]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds [1024 x i32], ptr @X, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp ne <4 x i32> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT:    [[TMP5:%.*]] = freeze <4 x i1> [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VECTOR_CHECK_EXIT:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP10:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP11:%.*]] = load i32, ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[TMP12:%.*]] = load i32, ptr [[TMP8]], align 4
+; CHECK-NEXT:    [[TMP13:%.*]] = load i32, ptr [[TMP9]], align 4
+; CHECK-NEXT:    [[TMP14:%.*]] = load i32, ptr [[TMP10]], align 4
+; CHECK-NEXT:    [[TMP15:%.*]] = insertelement <4 x i32> poison, i32 [[TMP11]], i64 0
+; CHECK-NEXT:    [[TMP16:%.*]] = insertelement <4 x i32> [[TMP15]], i32 [[TMP12]], i64 1
+; CHECK-NEXT:    [[TMP17:%.*]] = insertelement <4 x i32> [[TMP16]], i32 [[TMP13]], i64 2
+; CHECK-NEXT:    [[TMP18:%.*]] = insertelement <4 x i32> [[TMP17]], i32 [[TMP14]], i64 3
+; CHECK-NEXT:    [[TMP19:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP20:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP21:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP22:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP23:%.*]] = load i32, ptr [[TMP19]], align 4
+; CHECK-NEXT:    [[TMP24:%.*]] = load i32, ptr [[TMP20]], align 4
+; CHECK-NEXT:    [[TMP25:%.*]] = load i32, ptr [[TMP21]], align 4
+; CHECK-NEXT:    [[TMP26:%.*]] = load i32, ptr [[TMP22]], align 4
+; CHECK-NEXT:    [[TMP27:%.*]] = insertelement <4 x i32> poison, i32 [[TMP23]], i64 0
+; CHECK-NEXT:    [[TMP28:%.*]] = insertelement <4 x i32> [[TMP27]], i32 [[TMP24]], i64 1
+; CHECK-NEXT:    [[TMP29:%.*]] = insertelement <4 x i32> [[TMP28]], i32 [[TMP25]], i64 2
+; CHECK-NEXT:    [[TMP30:%.*]] = insertelement <4 x i32> [[TMP29]], i32 [[TMP26]], i64 3
+; CHECK-NEXT:    [[TMP31:%.*]] = add nsw <4 x i32> [[TMP18]], [[TMP30]]
+; CHECK-NEXT:    [[TMP32:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP33:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP34:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP35:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP36:%.*]] = extractelement <4 x i32> [[TMP31]], i64 0
+; CHECK-NEXT:    store i32 [[TMP36]], ptr [[TMP32]], align 4
+; CHECK-NEXT:    [[TMP37:%.*]] = extractelement <4 x i32> [[TMP31]], i64 1
+; CHECK-NEXT:    store i32 [[TMP37]], ptr [[TMP33]], align 4
+; CHECK-NEXT:    [[TMP38:%.*]] = extractelement <4 x i32> [[TMP31]], i64 2
+; CHECK-NEXT:    store i32 [[TMP38]], ptr [[TMP34]], align 4
+; CHECK-NEXT:    [[TMP39:%.*]] = extractelement <4 x i32> [[TMP31]], i64 3
+; CHECK-NEXT:    store i32 [[TMP39]], ptr [[TMP35]], align 4
+; CHECK-NEXT:    [[TMP40:%.*]] = getelementptr inbounds [1024 x i32], ptr @E, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP41:%.*]] = getelementptr inbounds [1024 x i32], ptr @E, i64 0, i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP42:%.*]] = getelementptr inbounds [1024 x i32], ptr @E, i64 0, i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP43:%.*]] = getelementptr inbounds [1024 x i32], ptr @E, i64 0, i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP44:%.*]] = load i32, ptr [[TMP40]], align 4
+; CHECK-NEXT:    [[TMP45:%.*]] = load i32, ptr [[TMP41]], align 4
+; CHECK-NEXT:    [[TMP46:%.*]] = load i32, ptr [[TMP42]], align 4
+; CHECK-NEXT:    [[TMP47:%.*]] = load i32, ptr [[TMP43]], align 4
+; CHECK-NEXT:    [[TMP48:%.*]] = insertelement <4 x i32> poison, i32 [[TMP44]], i64 0
+; CHECK-NEXT:    [[TMP49:%.*]] = insertelement <4 x i32> [[TMP48]], i32 [[TMP45]], i64 1
+; CHECK-NEXT:    [[TMP50:%.*]] = insertelement <4 x i32> [[TMP49]], i32 [[TMP46]], i64 2
+; CHECK-NEXT:    [[TMP51:%.*]] = insertelement <4 x i32> [[TMP50]], i32 [[TMP47]], i64 3
+; CHECK-NEXT:    [[TMP52:%.*]] = getelementptr inbounds [1024 x i32], ptr @F, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP53:%.*]] = getelementptr inbounds [1024 x i32], ptr @F, i64 0, i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP54:%.*]] = getelementptr inbounds [1024 x i32], ptr @F, i64 0, i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP55:%.*]] = getelementptr inbounds [1024 x i32], ptr @F, i64 0, i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP56:%.*]] = load i32, ptr [[TMP52]], align 4
+; CHECK-NEXT:    [[TMP57:%.*]] = load i32, ptr [[TMP53]], align 4
+; CHECK-NEXT:    [[TMP58:%.*]] = load i32, ptr [[TMP54]], align 4
+; CHECK-NEXT:    [[TMP59:%.*]] = load i32, ptr [[TMP55]], align 4
+; CHECK-NEXT:    [[TMP60:%.*]] = insertelement <4 x i32> poison, i32 [[TMP56]], i64 0
+; CHECK-NEXT:    [[TMP61:%.*]] = insertelement <4 x i32> [[TMP60]], i32 [[TMP57]], i64 1
+; CHECK-NEXT:    [[TMP62:%.*]] = insertelement <4 x i32> [[TMP61]], i32 [[TMP58]], i64 2
+; CHECK-NEXT:    [[TMP63:%.*]] = insertelement <4 x i32> [[TMP62]], i32 [[TMP59]], i64 3
+; CHECK-NEXT:    [[TMP64:%.*]] = add nsw <4 x i32> [[TMP51]], [[TMP63]]
+; CHECK-NEXT:    [[TMP65:%.*]] = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP66:%.*]] = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP67:%.*]] = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP68:%.*]] = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP69:%.*]] = extractelement <4 x i32> [[TMP64]], i64 0
+; CHECK-NEXT:    store i32 [[TMP69]], ptr [[TMP65]], align 4
+; CHECK-NEXT:    [[TMP70:%.*]] = extractelement <4 x i32> [[TMP64]], i64 1
+; CHECK-NEXT:    store i32 [[TMP70]], ptr [[TMP66]], align 4
+; CHECK-NEXT:    [[TMP71:%.*]] = extractelement <4 x i32> [[TMP64]], i64 2
+; CHECK-NEXT:    store i32 [[TMP71]], ptr [[TMP67]], align 4
+; CHECK-NEXT:    [[TMP72:%.*]] = extractelement <4 x i32> [[TMP64]], i64 3
+; CHECK-NEXT:    store i32 [[TMP72]], ptr [[TMP68]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP73:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT:    br i1 [[TMP73]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_CHECK]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[RET_LOOPEXIT:.*]]
+; CHECK:       [[VECTOR_CHECK_EXIT]]:
+; CHECK-NEXT:    br label %[[SCALAR_PH:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[INDEX]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[GB:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LB:%.*]] = load i32, ptr [[GB]], align 4
+; CHECK-NEXT:    [[GC:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LC:%.*]] = load i32, ptr [[GC]], align 4
+; CHECK-NEXT:    [[SUMA:%.*]] = add nsw i32 [[LB]], [[LC]]
+; CHECK-NEXT:    [[GA:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUMA]], ptr [[GA]], align 4
+; CHECK-NEXT:    [[GX:%.*]] = getelementptr inbounds [1024 x i32], ptr @X, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LX:%.*]] = load i32, ptr [[GX]], align 4
+; CHECK-NEXT:    [[CX:%.*]] = icmp ne i32 [[LX]], 0
+; CHECK-NEXT:    br i1 [[CX]], label %[[EXIT:.*]], label %[[IF_END:.*]]
+; CHECK:       [[IF_END]]:
+; CHECK-NEXT:    [[GE:%.*]] = getelementptr inbounds [1024 x i32], ptr @E, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LE:%.*]] = load i32, ptr [[GE]], align 4
+; CHECK-NEXT:    [[GF:%.*]] = getelementptr inbounds [1024 x i32], ptr @F, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LF:%.*]] = load i32, ptr [[GF]], align 4
+; CHECK-NEXT:    [[SUMD:%.*]] = add nsw i32 [[LE]], [[LF]]
+; CHECK-NEXT:    [[GD:%.*]] = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUMD]], ptr [[GD]], align 4
+; CHECK-NEXT:    br label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], 1024
+; CHECK-NEXT:    br i1 [[DONE]], label %[[RET_LOOPEXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[RET:.*]]
+; CHECK:       [[RET_LOOPEXIT]]:
+; CHECK-NEXT:    br label %[[RET]]
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %gb = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 %iv
+  %lb = load i32, ptr %gb, align 4
+  %gc = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 %iv
+  %lc = load i32, ptr %gc, align 4
+  %suma = add nsw i32 %lb, %lc
+  %ga = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 %iv
+  store i32 %suma, ptr %ga, align 4
+  %gx = getelementptr inbounds [1024 x i32], ptr @X, i64 0, i64 %iv
+  %lx = load i32, ptr %gx, align 4
+  %cx = icmp ne i32 %lx, 0
+  br i1 %cx, label %exit, label %if.end
+
+if.end:
+  %ge = getelementptr inbounds [1024 x i32], ptr @E, i64 0, i64 %iv
+  %le = load i32, ptr %ge, align 4
+  %gf = getelementptr inbounds [1024 x i32], ptr @F, i64 0, i64 %iv
+  %lf = load i32, ptr %gf, align 4
+  %sumd = add nsw i32 %le, %lf
+  %gd = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 %iv
+  store i32 %sumd, ptr %gd, align 4
+  br label %latch
+
+latch:
+  %iv.next = add nuw nsw i64 %iv, 1
+  %done = icmp eq i64 %iv.next, 1024
+  br i1 %done, label %ret, label %for.body
+
+exit:
+  br label %ret
+
+ret:
+  ret void
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/check-first-multi-exit-cascade.ll b/llvm/test/Transforms/LoopVectorize/check-first-multi-exit-cascade.ll
new file mode 100644
index 0000000000000..a53ec63c14815
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/check-first-multi-exit-cascade.ll
@@ -0,0 +1,133 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 -enable-check-first-early-exit-vectorization | FileCheck %s
+
+ at A = global [1024 x i32] zeroinitializer
+ at B = global [1024 x i32] zeroinitializer
+ at D = global [1024 x i32] zeroinitializer
+
+define i32 @multi_exit_cascade() {
+; CHECK-LABEL: define i32 @multi_exit_cascade() {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    br label %[[VECTOR_CHECK:.*]]
+; CHECK:       [[VECTOR_CHECK]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY:.*]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq <4 x i32> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT:    [[TMP5:%.*]] = freeze <4 x i1> [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VECTOR_CHECK_EXIT:.*]], label %[[VECTOR_CHECK3:.*]]
+; CHECK:       [[VECTOR_CHECK3]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD4:%.*]] = load <4 x i32>, ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[TMP8:%.*]] = icmp eq <4 x i32> [[WIDE_LOAD4]], zeroinitializer
+; CHECK-NEXT:    [[TMP9:%.*]] = freeze <4 x i1> [[TMP8]]
+; CHECK-NEXT:    [[TMP10:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP9]])
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[VECTOR_CHECK_EXIT]], label %[[VECTOR_BODY]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[TMP11:%.*]] = add <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD4]]
+; CHECK-NEXT:    [[TMP12:%.*]] = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP13:%.*]] = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP14:%.*]] = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP15:%.*]] = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP16:%.*]] = extractelement <4 x i32> [[TMP11]], i64 0
+; CHECK-NEXT:    store i32 [[TMP16]], ptr [[TMP12]], align 4
+; CHECK-NEXT:    [[TMP17:%.*]] = extractelement <4 x i32> [[TMP11]], i64 1
+; CHECK-NEXT:    store i32 [[TMP17]], ptr [[TMP13]], align 4
+; CHECK-NEXT:    [[TMP18:%.*]] = extractelement <4 x i32> [[TMP11]], i64 2
+; CHECK-NEXT:    store i32 [[TMP18]], ptr [[TMP14]], align 4
+; CHECK-NEXT:    [[TMP19:%.*]] = extractelement <4 x i32> [[TMP11]], i64 3
+; CHECK-NEXT:    store i32 [[TMP19]], ptr [[TMP15]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP20:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT:    br i1 [[TMP20]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_CHECK]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[RET_N:.*]]
+; CHECK:       [[VECTOR_CHECK_EXIT]]:
+; CHECK-NEXT:    br label %[[SCALAR_PH:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[INDEX]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[GA:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LA:%.*]] = load i32, ptr [[GA]], align 4
+; CHECK-NEXT:    [[CA:%.*]] = icmp eq i32 [[LA]], 0
+; CHECK-NEXT:    br i1 [[CA]], label %[[EXIT0:.*]], label %[[CHECK1:.*]]
+; CHECK:       [[CHECK1]]:
+; CHECK-NEXT:    [[GB:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LB:%.*]] = load i32, ptr [[GB]], align 4
+; CHECK-NEXT:    [[CB:%.*]] = icmp eq i32 [[LB]], 0
+; CHECK-NEXT:    br i1 [[CB]], label %[[EXIT1:.*]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[SUM:%.*]] = add i32 [[LA]], [[LB]]
+; CHECK-NEXT:    [[GD:%.*]] = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUM]], ptr [[GD]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], 1024
+; CHECK-NEXT:    br i1 [[DONE]], label %[[RET_N]], label %[[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[EXIT0]]:
+; CHECK-NEXT:    [[IV_LCSSA:%.*]] = phi i64 [ [[IV]], %[[FOR_BODY]] ]
+; CHECK-NEXT:    [[T0:%.*]] = trunc i64 [[IV_LCSSA]] to i32
+; CHECK-NEXT:    br label %[[RET:.*]]
+; CHECK:       [[EXIT1]]:
+; CHECK-NEXT:    [[IV_LCSSA1:%.*]] = phi i64 [ [[IV]], %[[CHECK1]] ]
+; CHECK-NEXT:    [[T1:%.*]] = trunc i64 [[IV_LCSSA1]] to i32
+; CHECK-NEXT:    [[NEG:%.*]] = sub i32 0, [[T1]]
+; CHECK-NEXT:    br label %[[RET]]
+; CHECK:       [[RET_N]]:
+; CHECK-NEXT:    br label %[[RET]]
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    [[R:%.*]] = phi i32 [ [[T0]], %[[EXIT0]] ], [ [[NEG]], %[[EXIT1]] ], [ 1024, %[[RET_N]] ]
+; CHECK-NEXT:    ret i32 [[R]]
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %ga = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 %iv
+  %la = load i32, ptr %ga, align 4
+  %ca = icmp eq i32 %la, 0
+  br i1 %ca, label %exit0, label %check1
+
+check1:
+  %gb = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 %iv
+  %lb = load i32, ptr %gb, align 4
+  %cb = icmp eq i32 %lb, 0
+  br i1 %cb, label %exit1, label %latch
+
+latch:
+  %sum = add i32 %la, %lb
+  %gd = getelementptr inbounds [1024 x i32], ptr @D, i64 0, i64 %iv
+  store i32 %sum, ptr %gd, align 4
+  %iv.next = add nuw nsw i64 %iv, 1
+  %done = icmp eq i64 %iv.next, 1024
+  br i1 %done, label %ret.n, label %for.body
+
+exit0:
+  %t0 = trunc i64 %iv to i32
+  br label %ret
+
+exit1:
+  %t1 = trunc i64 %iv to i32
+  %neg = sub i32 0, %t1
+  br label %ret
+
+ret.n:
+  br label %ret
+
+ret:
+  %r = phi i32 [ %t0, %exit0 ], [ %neg, %exit1 ], [ 1024, %ret.n ]
+  ret i32 %r
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/check-first-single-exit.ll b/llvm/test/Transforms/LoopVectorize/check-first-single-exit.ll
new file mode 100644
index 0000000000000..cef8d221d4b3d
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/check-first-single-exit.ll
@@ -0,0 +1,136 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 -enable-check-first-early-exit-vectorization | FileCheck %s
+
+;   for (i = 0; i < 1024; ++i) {
+;     A[i] = B[i] + C[i];
+;     if (X[i])
+;       break;
+;   }
+
+ at A = global [1024 x i32] zeroinitializer
+ at B = global [1024 x i32] zeroinitializer
+ at C = global [1024 x i32] zeroinitializer
+ at X = global [1024 x i32] zeroinitializer
+
+define void @single_exit() {
+; CHECK-LABEL: define void @single_exit() {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    br label %[[VECTOR_CHECK:.*]]
+; CHECK:       [[VECTOR_CHECK]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY:.*]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds [1024 x i32], ptr @X, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp ne <4 x i32> [[WIDE_LOAD]], zeroinitializer
+; CHECK-NEXT:    [[TMP5:%.*]] = freeze <4 x i1> [[TMP4]]
+; CHECK-NEXT:    [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
+; CHECK-NEXT:    br i1 [[TMP6]], label %[[VECTOR_CHECK_EXIT:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[TMP7:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP8:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP10:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP11:%.*]] = load i32, ptr [[TMP7]], align 4
+; CHECK-NEXT:    [[TMP12:%.*]] = load i32, ptr [[TMP8]], align 4
+; CHECK-NEXT:    [[TMP13:%.*]] = load i32, ptr [[TMP9]], align 4
+; CHECK-NEXT:    [[TMP14:%.*]] = load i32, ptr [[TMP10]], align 4
+; CHECK-NEXT:    [[TMP15:%.*]] = insertelement <4 x i32> poison, i32 [[TMP11]], i64 0
+; CHECK-NEXT:    [[TMP16:%.*]] = insertelement <4 x i32> [[TMP15]], i32 [[TMP12]], i64 1
+; CHECK-NEXT:    [[TMP17:%.*]] = insertelement <4 x i32> [[TMP16]], i32 [[TMP13]], i64 2
+; CHECK-NEXT:    [[TMP18:%.*]] = insertelement <4 x i32> [[TMP17]], i32 [[TMP14]], i64 3
+; CHECK-NEXT:    [[TMP19:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP20:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP21:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP22:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP23:%.*]] = load i32, ptr [[TMP19]], align 4
+; CHECK-NEXT:    [[TMP24:%.*]] = load i32, ptr [[TMP20]], align 4
+; CHECK-NEXT:    [[TMP25:%.*]] = load i32, ptr [[TMP21]], align 4
+; CHECK-NEXT:    [[TMP26:%.*]] = load i32, ptr [[TMP22]], align 4
+; CHECK-NEXT:    [[TMP27:%.*]] = insertelement <4 x i32> poison, i32 [[TMP23]], i64 0
+; CHECK-NEXT:    [[TMP28:%.*]] = insertelement <4 x i32> [[TMP27]], i32 [[TMP24]], i64 1
+; CHECK-NEXT:    [[TMP29:%.*]] = insertelement <4 x i32> [[TMP28]], i32 [[TMP25]], i64 2
+; CHECK-NEXT:    [[TMP30:%.*]] = insertelement <4 x i32> [[TMP29]], i32 [[TMP26]], i64 3
+; CHECK-NEXT:    [[TMP31:%.*]] = add nsw <4 x i32> [[TMP18]], [[TMP30]]
+; CHECK-NEXT:    [[TMP32:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP33:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP34:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP35:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP36:%.*]] = extractelement <4 x i32> [[TMP31]], i64 0
+; CHECK-NEXT:    store i32 [[TMP36]], ptr [[TMP32]], align 4
+; CHECK-NEXT:    [[TMP37:%.*]] = extractelement <4 x i32> [[TMP31]], i64 1
+; CHECK-NEXT:    store i32 [[TMP37]], ptr [[TMP33]], align 4
+; CHECK-NEXT:    [[TMP38:%.*]] = extractelement <4 x i32> [[TMP31]], i64 2
+; CHECK-NEXT:    store i32 [[TMP38]], ptr [[TMP34]], align 4
+; CHECK-NEXT:    [[TMP39:%.*]] = extractelement <4 x i32> [[TMP31]], i64 3
+; CHECK-NEXT:    store i32 [[TMP39]], ptr [[TMP35]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP40:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT:    br i1 [[TMP40]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_CHECK]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[RET_LOOPEXIT:.*]]
+; CHECK:       [[VECTOR_CHECK_EXIT]]:
+; CHECK-NEXT:    br label %[[SCALAR_PH:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[INDEX]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[GB:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LB:%.*]] = load i32, ptr [[GB]], align 4
+; CHECK-NEXT:    [[GC:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LC:%.*]] = load i32, ptr [[GC]], align 4
+; CHECK-NEXT:    [[SUMA:%.*]] = add nsw i32 [[LB]], [[LC]]
+; CHECK-NEXT:    [[GA:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUMA]], ptr [[GA]], align 4
+; CHECK-NEXT:    [[GX:%.*]] = getelementptr inbounds [1024 x i32], ptr @X, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LX:%.*]] = load i32, ptr [[GX]], align 4
+; CHECK-NEXT:    [[CX:%.*]] = icmp ne i32 [[LX]], 0
+; CHECK-NEXT:    br i1 [[CX]], label %[[EXIT:.*]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], 1024
+; CHECK-NEXT:    br i1 [[DONE]], label %[[RET_LOOPEXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[RET:.*]]
+; CHECK:       [[RET_LOOPEXIT]]:
+; CHECK-NEXT:    br label %[[RET]]
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %gb = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 %iv
+  %lb = load i32, ptr %gb, align 4
+  %gc = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 %iv
+  %lc = load i32, ptr %gc, align 4
+  %suma = add nsw i32 %lb, %lc
+  %ga = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 %iv
+  store i32 %suma, ptr %ga, align 4
+  %gx = getelementptr inbounds [1024 x i32], ptr @X, i64 0, i64 %iv
+  %lx = load i32, ptr %gx, align 4
+  %cx = icmp ne i32 %lx, 0
+  br i1 %cx, label %exit, label %latch
+
+latch:
+  %iv.next = add nuw nsw i64 %iv, 1
+  %done = icmp eq i64 %iv.next, 1024
+  br i1 %done, label %ret, label %for.body
+
+exit:
+  br label %ret
+
+ret:
+  ret void
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.
diff --git a/llvm/test/Transforms/LoopVectorize/check-first-state-update.ll b/llvm/test/Transforms/LoopVectorize/check-first-state-update.ll
new file mode 100644
index 0000000000000..bc4b6c7356f09
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/check-first-state-update.ll
@@ -0,0 +1,109 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -S < %s -p loop-vectorize -force-vector-width=4 -enable-check-first-early-exit-vectorization | FileCheck %s
+
+;   for (i = 0; i < 1024; i++) {
+;     A[i] = B[i] + C[i];
+;     if (A[i])
+;       break;
+;   }
+
+ at A = global [1024 x i32] zeroinitializer
+ at B = global [1024 x i32] zeroinitializer
+ at C = global [1024 x i32] zeroinitializer
+
+define void @state_update() {
+; CHECK-LABEL: define void @state_update() {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    br label %[[VECTOR_CHECK:.*]]
+; CHECK:       [[VECTOR_CHECK]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY:.*]] ]
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[INDEX]], 1
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[INDEX]], 2
+; CHECK-NEXT:    [[TMP2:%.*]] = add i64 [[INDEX]], 3
+; CHECK-NEXT:    [[TMP3:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP3]], align 4
+; CHECK-NEXT:    [[TMP4:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD1:%.*]] = load <4 x i32>, ptr [[TMP4]], align 4
+; CHECK-NEXT:    [[TMP5:%.*]] = add nsw <4 x i32> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp ne <4 x i32> [[TMP5]], zeroinitializer
+; CHECK-NEXT:    [[TMP7:%.*]] = freeze <4 x i1> [[TMP6]]
+; CHECK-NEXT:    [[TMP8:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP7]])
+; CHECK-NEXT:    br i1 [[TMP8]], label %[[VECTOR_CHECK_EXIT:.*]], label %[[VECTOR_BODY]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[TMP9:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[INDEX]]
+; CHECK-NEXT:    [[TMP10:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[TMP0]]
+; CHECK-NEXT:    [[TMP11:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[TMP1]]
+; CHECK-NEXT:    [[TMP12:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[TMP2]]
+; CHECK-NEXT:    [[TMP13:%.*]] = extractelement <4 x i32> [[TMP5]], i64 0
+; CHECK-NEXT:    store i32 [[TMP13]], ptr [[TMP9]], align 4
+; CHECK-NEXT:    [[TMP14:%.*]] = extractelement <4 x i32> [[TMP5]], i64 1
+; CHECK-NEXT:    store i32 [[TMP14]], ptr [[TMP10]], align 4
+; CHECK-NEXT:    [[TMP15:%.*]] = extractelement <4 x i32> [[TMP5]], i64 2
+; CHECK-NEXT:    store i32 [[TMP15]], ptr [[TMP11]], align 4
+; CHECK-NEXT:    [[TMP16:%.*]] = extractelement <4 x i32> [[TMP5]], i64 3
+; CHECK-NEXT:    store i32 [[TMP16]], ptr [[TMP12]], align 4
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP17:%.*]] = icmp eq i64 [[INDEX_NEXT]], 1024
+; CHECK-NEXT:    br i1 [[TMP17]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_CHECK]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br label %[[RET_LOOPEXIT:.*]]
+; CHECK:       [[VECTOR_CHECK_EXIT]]:
+; CHECK-NEXT:    br label %[[SCALAR_PH:.*]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    br label %[[FOR_BODY:.*]]
+; CHECK:       [[FOR_BODY]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[INDEX]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[GB:%.*]] = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LB:%.*]] = load i32, ptr [[GB]], align 4
+; CHECK-NEXT:    [[GC:%.*]] = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 [[IV]]
+; CHECK-NEXT:    [[LC:%.*]] = load i32, ptr [[GC]], align 4
+; CHECK-NEXT:    [[SUMA:%.*]] = add nsw i32 [[LB]], [[LC]]
+; CHECK-NEXT:    [[GA:%.*]] = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 [[IV]]
+; CHECK-NEXT:    store i32 [[SUMA]], ptr [[GA]], align 4
+; CHECK-NEXT:    [[CA:%.*]] = icmp ne i32 [[SUMA]], 0
+; CHECK-NEXT:    br i1 [[CA]], label %[[EXIT:.*]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[IV_NEXT]], 1024
+; CHECK-NEXT:    br i1 [[DONE]], label %[[RET_LOOPEXIT]], label %[[FOR_BODY]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[RET:.*]]
+; CHECK:       [[RET_LOOPEXIT]]:
+; CHECK-NEXT:    br label %[[RET]]
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %for.body
+
+for.body:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %gb = getelementptr inbounds [1024 x i32], ptr @B, i64 0, i64 %iv
+  %lb = load i32, ptr %gb, align 4
+  %gc = getelementptr inbounds [1024 x i32], ptr @C, i64 0, i64 %iv
+  %lc = load i32, ptr %gc, align 4
+  %suma = add nsw i32 %lb, %lc
+  %ga = getelementptr inbounds [1024 x i32], ptr @A, i64 0, i64 %iv
+  store i32 %suma, ptr %ga, align 4
+  %ca = icmp ne i32 %suma, 0
+  br i1 %ca, label %exit, label %latch
+
+latch:
+  %iv.next = add nuw nsw i64 %iv, 1
+  %done = icmp eq i64 %iv.next, 1024
+  br i1 %done, label %ret, label %for.body
+
+exit:
+  br label %ret
+
+ret:
+  ret void
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[LOOP3]] = distinct !{[[LOOP3]], [[META2]], [[META1]]}
+;.



More information about the llvm-commits mailing list