[llvm] [Transforms][Utils] Add LoopSplit for iteration-space loop splitting (PR #217232)

Ashutosh Nema via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 22 03:54:20 PDT 2026


https://github.com/nema-ashutosh updated https://github.com/llvm/llvm-project/pull/217232

>From d2bc3ce15cb69ef248a29b6d898049fd9b01bf3f Mon Sep 17 00:00:00 2001
From: Ashutosh Nema <ashu1212 at gmail.com>
Date: Wed, 19 Aug 2026 13:31:09 +0530
Subject: [PATCH 1/9] [Transforms][Utils] Add LoopSplit for iteration-space
 loop splitting

Split a counted loop into a chain of per-partition sub-loops covering
contiguous slices of the iteration space, each guarded by an entry check
with its latch clamped to its slice.

Reintroduce a reduced version of LoopSplitUtils. Handles a unit-step
integer induction in either direction and both signed and unsigned
orderings; loops with exit values or loop-carried values are rejected.

Driven for testing by opt -passes=loop-split.
---
 .../include/llvm/Transforms/Utils/LoopSplit.h | 132 +++
 .../llvm/Transforms/Utils/LoopSplitPass.h     |  30 +
 llvm/lib/Passes/PassBuilder.cpp               |   1 +
 llvm/lib/Passes/PassRegistry.def              |   1 +
 llvm/lib/Transforms/Utils/CMakeLists.txt      |   2 +
 llvm/lib/Transforms/Utils/LoopSplit.cpp       | 527 +++++++++++
 llvm/lib/Transforms/Utils/LoopSplitPass.cpp   | 143 +++
 llvm/test/Transforms/LoopSplit/basic.ll       | 263 ++++++
 .../Transforms/LoopSplit/branch-weights.ll    |  70 ++
 .../LoopSplit/constant-trip-count.ll          | 153 +++
 llvm/test/Transforms/LoopSplit/descending.ll  | 170 ++++
 llvm/test/Transforms/LoopSplit/loop-depth.ll  | 638 +++++++++++++
 .../LoopSplit/multiple-partitions.ll          | 155 +++
 .../LoopSplit/oversized-split-offset.ll       |  87 ++
 .../Transforms/LoopSplit/saturating-latch.ll  | 222 +++++
 .../Transforms/LoopSplit/sequential-loops.ll  | 107 +++
 .../Transforms/LoopSplit/unsupported-forms.ll | 879 ++++++++++++++++++
 .../LoopSplit/wrapping-iteration-space.ll     | 139 +++
 18 files changed, 3719 insertions(+)
 create mode 100644 llvm/include/llvm/Transforms/Utils/LoopSplit.h
 create mode 100644 llvm/include/llvm/Transforms/Utils/LoopSplitPass.h
 create mode 100644 llvm/lib/Transforms/Utils/LoopSplit.cpp
 create mode 100644 llvm/lib/Transforms/Utils/LoopSplitPass.cpp
 create mode 100644 llvm/test/Transforms/LoopSplit/basic.ll
 create mode 100644 llvm/test/Transforms/LoopSplit/branch-weights.ll
 create mode 100644 llvm/test/Transforms/LoopSplit/constant-trip-count.ll
 create mode 100644 llvm/test/Transforms/LoopSplit/descending.ll
 create mode 100644 llvm/test/Transforms/LoopSplit/loop-depth.ll
 create mode 100644 llvm/test/Transforms/LoopSplit/multiple-partitions.ll
 create mode 100644 llvm/test/Transforms/LoopSplit/oversized-split-offset.ll
 create mode 100644 llvm/test/Transforms/LoopSplit/saturating-latch.ll
 create mode 100644 llvm/test/Transforms/LoopSplit/sequential-loops.ll
 create mode 100644 llvm/test/Transforms/LoopSplit/unsupported-forms.ll
 create mode 100644 llvm/test/Transforms/LoopSplit/wrapping-iteration-space.ll

diff --git a/llvm/include/llvm/Transforms/Utils/LoopSplit.h b/llvm/include/llvm/Transforms/Utils/LoopSplit.h
new file mode 100644
index 0000000000000..b075d76b843ad
--- /dev/null
+++ b/llvm/include/llvm/Transforms/Utils/LoopSplit.h
@@ -0,0 +1,132 @@
+//===- LoopSplit.h - Split a loop's iteration space -------------*- C++ -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// Splits a counted loop's iteration space into a chain of per-partition
+// sub-loops. See LoopSplit.cpp for the structure produced.
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef LLVM_TRANSFORMS_UTILS_LOOPSPLIT_H
+#define LLVM_TRANSFORMS_UTILS_LOOPSPLIT_H
+
+#include "llvm/ADT/SmallVector.h"
+#include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Support/Compiler.h"
+
+namespace llvm {
+
+class DominatorTree;
+class SCEV;
+class SCEVExpander;
+class ScalarEvolution;
+
+/// Splits a counted loop into a chain of per-partition sub-loops.
+///
+/// Usage:
+/// \code
+///   LoopSplit LS(L, LI, SE, DT);
+///   if (!LS.isLegal())
+///     return false;
+///   LS.addPartition(S0, E0);   // one call per partition, in order
+///   LS.addPartition(S1, E1);
+///   LS.split();
+/// \endcode
+class LoopSplit {
+public:
+  LLVM_ABI LoopSplit(Loop *L, LoopInfo *LI, ScalarEvolution *SE,
+                     DominatorTree *DT)
+      : L(L), LI(LI), SE(SE), DT(DT) {}
+
+  /// Analyze \p L and return true if it is a counted loop this utility can
+  /// split: a bottom-tested single-exit loop in LCSSA form with dedicated
+  /// exits, no loop-carried and no escaping values, a unique unit-step integer
+  /// induction, and a computable trip count that cannot wrap. Must succeed
+  /// before split().
+  LLVM_ABI bool isLegal();
+
+  /// Return the loop's induction variable. Valid only after isLegal() succeeds.
+  LLVM_ABI PHINode *getInductionVariable() const {
+    return L->getInductionVariable(*SE);
+  }
+
+  /// The induction value on the last iteration, which the final partition must
+  /// end at. Valid only after isLegal() succeeds.
+  LLVM_ABI const SCEV *getInductionEnd() const { return InductionEnd; }
+
+  /// Append an inclusive partition range [Start, End] in iteration order.
+  /// Partitions must tile the whole space: first Start = induction start, each
+  /// later Start = previous End +/- step, last End = induction end (desc: S >=
+  /// E).
+  ///
+  /// Both bounds must have the induction type and be loop-invariant. They must
+  /// also stay within the iteration space, extended by the one step past its
+  /// start that an empty partition needs; isLegal() has proven that much
+  /// representable. Reaching further wraps past TYPE_MAX/MIN/0 into a bound
+  /// that still looks in range, which silently miscompiles. See LoopSplit.cpp
+  /// for the rationale.
+  LLVM_ABI void addPartition(const SCEV *Start, const SCEV *End);
+
+  LLVM_ABI unsigned getNumPartitions() const { return Partitions.size(); }
+
+  /// Perform the split. Requires a successful isLegal() and at least two
+  /// partitions. Returns true if the loop was rewritten.
+  LLVM_ABI bool split();
+
+private:
+  /// Everything known about one partition: the caller-supplied range plus the
+  /// state split() derives. Indexed by partition number in \c Partitions.
+  struct PartitionInfo {
+    PartitionInfo(const SCEV *StartExpr, const SCEV *EndExpr)
+        : StartExpr(StartExpr), EndExpr(EndExpr) {}
+
+    // Set by addPartition() before split():
+    const SCEV *StartExpr; // inclusive iteration range [Start, End].
+    const SCEV *EndExpr;
+
+    // Filled in by split():
+    Value *StartVal = nullptr; // expanded start.
+    Value *SelEnd = nullptr;   // clamped end min(End, indEnd).
+    BasicBlock *GuardBlock = nullptr;
+    BasicBlock *Preheader = nullptr;
+    BasicBlock *Exit = nullptr;
+    Loop *SubLoop = nullptr;
+    PHINode *IndPHI = nullptr; // this partition's induction variable.
+  };
+
+  /// Per-split() scratch threaded through the phase helpers: the blocks the
+  /// transform creates. A pure transform internal, so it is defined in the
+  /// implementation file.
+  struct SplitState;
+
+  Loop *L;
+  LoopInfo *LI;
+  ScalarEvolution *SE;
+  DominatorTree *DT;
+
+  // Induction analysis, populated by isLegal().
+  const SCEV *InductionEnd = nullptr; // value on the last iteration.
+  bool InductionIsSigned = false;     // iteration ordering signedness.
+  bool Descending = false;            // step is -1 (the loop counts down).
+
+  /// One record per partition, in add order.
+  SmallVector<PartitionInfo, 4> Partitions;
+
+  // split() phase helpers, run in order; each is documented at its definition.
+  /// Split the final exit off the loop exit block.
+  void splitFinalExit(SplitState &S);
+  /// Expand each partition's start and clamped end into the entry guard.
+  void expandPartitionBounds(SplitState &S, SCEVExpander &Expander);
+  /// Clone each later partition's sub-loop and create its guard/exit.
+  void clonePartitions(SplitState &S);
+  /// Emit each guard, clamp each latch, and chain the partitions.
+  void chainPartitions(SplitState &S);
+};
+
+} // namespace llvm
+
+#endif // LLVM_TRANSFORMS_UTILS_LOOPSPLIT_H
diff --git a/llvm/include/llvm/Transforms/Utils/LoopSplitPass.h b/llvm/include/llvm/Transforms/Utils/LoopSplitPass.h
new file mode 100644
index 0000000000000..14dacfa8b5b7f
--- /dev/null
+++ b/llvm/include/llvm/Transforms/Utils/LoopSplitPass.h
@@ -0,0 +1,30 @@
+//===- LoopSplitPass.h - Test driver for LoopSplit --------------*- C++ -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// A command-line driven pass used to exercise the LoopSplit utility from `opt`.
+// The split points are provided via the -loop-split-points option as iteration
+// offsets relative to the induction start.
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef LLVM_TRANSFORMS_UTILS_LOOPSPLITPASS_H
+#define LLVM_TRANSFORMS_UTILS_LOOPSPLITPASS_H
+
+#include "llvm/IR/PassManager.h"
+#include "llvm/Support/Compiler.h"
+
+namespace llvm {
+
+class LoopSplitPass : public PassInfoMixin<LoopSplitPass> {
+public:
+  LLVM_ABI PreservedAnalyses run(Function &F, FunctionAnalysisManager &AM);
+};
+
+} // namespace llvm
+
+#endif // LLVM_TRANSFORMS_UTILS_LOOPSPLITPASS_H
diff --git a/llvm/lib/Passes/PassBuilder.cpp b/llvm/lib/Passes/PassBuilder.cpp
index 858c76706427e..cc8dee8f08a18 100644
--- a/llvm/lib/Passes/PassBuilder.cpp
+++ b/llvm/lib/Passes/PassBuilder.cpp
@@ -394,6 +394,7 @@
 #include "llvm/Transforms/Utils/InstructionNamer.h"
 #include "llvm/Transforms/Utils/LibCallsShrinkWrap.h"
 #include "llvm/Transforms/Utils/LoopSimplify.h"
+#include "llvm/Transforms/Utils/LoopSplitPass.h"
 #include "llvm/Transforms/Utils/LoopVersioning.h"
 #include "llvm/Transforms/Utils/LowerCommentStringPass.h"
 #include "llvm/Transforms/Utils/LowerGlobalDtors.h"
diff --git a/llvm/lib/Passes/PassRegistry.def b/llvm/lib/Passes/PassRegistry.def
index 33c9e19988d7a..57687fe366a9c 100644
--- a/llvm/lib/Passes/PassRegistry.def
+++ b/llvm/lib/Passes/PassRegistry.def
@@ -478,6 +478,7 @@ FUNCTION_PASS("loop-fusion", LoopFusePass())
 FUNCTION_PASS("loop-load-elim", LoopLoadEliminationPass())
 FUNCTION_PASS("loop-simplify", LoopSimplifyPass())
 FUNCTION_PASS("loop-sink", LoopSinkPass())
+FUNCTION_PASS("loop-split", LoopSplitPass())
 FUNCTION_PASS("loop-versioning", LoopVersioningPass())
 FUNCTION_PASS("lower-atomic", LowerAtomicPass())
 FUNCTION_PASS("lower-constant-intrinsics", LowerConstantIntrinsicsPass())
diff --git a/llvm/lib/Transforms/Utils/CMakeLists.txt b/llvm/lib/Transforms/Utils/CMakeLists.txt
index 3b7a134b49e2e..d2f61c7ae4c5d 100644
--- a/llvm/lib/Transforms/Utils/CMakeLists.txt
+++ b/llvm/lib/Transforms/Utils/CMakeLists.txt
@@ -50,6 +50,8 @@ add_llvm_component_library(LLVMTransformUtils
   LoopPeel.cpp
   LoopRotationUtils.cpp
   LoopSimplify.cpp
+  LoopSplit.cpp
+  LoopSplitPass.cpp
   LoopUnroll.cpp
   LoopUnrollAndJam.cpp
   LoopUnrollRuntime.cpp
diff --git a/llvm/lib/Transforms/Utils/LoopSplit.cpp b/llvm/lib/Transforms/Utils/LoopSplit.cpp
new file mode 100644
index 0000000000000..b4eaf7de3fdd0
--- /dev/null
+++ b/llvm/lib/Transforms/Utils/LoopSplit.cpp
@@ -0,0 +1,527 @@
+//===- LoopSplit.cpp - Split a loop's iteration space ---------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// Splits a counted loop's iteration space into a chain of per-partition
+// sub-loops. See LoopSplit.h for the high-level usage guidelines.
+//
+// Structure produced for partitions [S0,E0], [S1,E1], ... where E is the loop's
+// last iteration and each clamped end sel_i is min(E_i, E), or max descending:
+//
+//   guard0:                            ; every S_i and sel_i is computed here
+//     if (S0 <= sel0) goto preheader0 else goto guard1
+//   loop0: ...                         ; latch iterates while i < sel0
+//   exit0 -> guard1
+//   guard1:
+//     if (S1 <= sel1) goto preheader1 else goto guard2
+//   loop1: ...                         ; latch iterates while i < sel1
+//   exit1 -> guard2
+//     ...
+//   final.exit:
+//
+// Each guard holds the "S_i <= sel_i" check and skips an empty partition by
+// falling through to the next guard. All S_i/sel_i are materialized once in
+// guard0, and the end clamp keeps the "runs at least once" iteration in the
+// right partition.
+//
+// The latch keeps iterating while the value the next iteration would use is
+// still in the partition. That is written as the strict "i < sel_i" on the
+// induction PHI rather than "i + 1 <= sel_i" on the step value; the two agree
+// because isLegal() has established that the space does not wrap, and the
+// strict form never forms i + 1, so it remains a real test even when sel_i is
+// the last value of the type, where the inclusive one would be a tautology and
+// the partition would never exit.
+//
+// A descending (step -1) loop uses the same structure mirrored: partitions run
+// high-to-low and the clamp and predicates flip (>=/>).
+//
+// Usage guidelines:
+//  - Caller bounds must not wrap the induction type. The clamp absorbs a bound
+//    past the runtime trip count, and isLegal() reserves the one step past the
+//    induction start that an empty partition needs, but a bound reaching any
+//    further wraps in the bound arithmetic and cannot be repaired here.
+//  - Bounds must be loop-invariant: they are expanded in guard0, so a bound
+//    depending on a value defined inside the loop cannot be placed.
+//  - The partitions must tile the original iteration space exactly -- same
+//    iterations, same order -- so the split preserves program behaviour.
+//
+// The transform is structural: inside a partition it only seeds the induction
+// PHI with that partition's start and replaces the latch test. It never
+// rebuilds a value that flows between partitions, so no SSA reconstruction is
+// needed.
+//
+// Not yet supported, and rejected by isLegal(): loop-carried values, values
+// that escape the loop (exit values), non-unit and non-integer inductions,
+// top-tested loops, and multiple exits. Also rejected is an induction start at
+// the extreme of the iteration direction, which leaves nowhere to put a
+// boundary. An induction *end* at that extreme is fine, because the latch stays
+// strict.
+//
+//===----------------------------------------------------------------------===//
+
+#include "llvm/Transforms/Utils/LoopSplit.h"
+#include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/ScalarEvolution.h"
+#include "llvm/Analysis/ScalarEvolutionExpressions.h"
+#include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
+#include "llvm/IR/BasicBlock.h"
+#include "llvm/IR/Dominators.h"
+#include "llvm/IR/Function.h"
+#include "llvm/IR/IRBuilder.h"
+#include "llvm/IR/Instructions.h"
+#include "llvm/IR/ProfDataUtils.h"
+#include "llvm/Support/Debug.h"
+#include "llvm/Transforms/Utils/BasicBlockUtils.h"
+#include "llvm/Transforms/Utils/Cloning.h"
+#include "llvm/Transforms/Utils/LoopUtils.h"
+#include "llvm/Transforms/Utils/ScalarEvolutionExpander.h"
+#include "llvm/Transforms/Utils/ValueMapper.h"
+#include <optional>
+
+using namespace llvm;
+using namespace llvm::SCEVPatternMatch;
+
+#define DEBUG_TYPE "loop-split"
+
+//===----------------------------------------------------------------------===//
+// LoopSplit - construction, partition list, induction analysis
+//===----------------------------------------------------------------------===//
+
+/// Per-split() scratch shared by the phase helpers; lives for one split() call.
+/// Everything derived from the induction lives on LoopSplit itself, filled in
+/// by isLegal(); this holds only what the transform creates.
+struct LoopSplit::SplitState {
+  // Partition 0 reuses the original loop's preheader, exit, and entry guard;
+  // those blocks live in Partitions[0] rather than being duplicated here.
+  BasicBlock *FinalExit = nullptr; // where the partition chain converges.
+  Loop *OuterLoop = nullptr;       // parent of the new blocks, if any.
+  PHINode *Induction = nullptr;    // the loop's induction variable.
+};
+
+// Record a new partition with the given inclusive iteration range.
+void LoopSplit::addPartition(const SCEV *Start, const SCEV *End) {
+  assert(InductionEnd && "addPartition() requires a successful isLegal()");
+  // The bounds are combined with the induction end and expanded in its type. A
+  // mismatch would otherwise surface either as a bare "Operand types don't
+  // match!" from inside ScalarEvolution, or worse, as a silent cast.
+  assert(Start->getType() == InductionEnd->getType() &&
+         End->getType() == InductionEnd->getType() &&
+         "partition bounds must have the induction type");
+  Partitions.emplace_back(Start, End);
+}
+
+// Return the induction's add-recurrence, or null unless the induction is an
+// integer with a unit step that the latch compares.
+static const SCEVAddRecExpr *analyzeInduction(Loop *L, ScalarEvolution *SE) {
+  ICmpInst *LatchCmp = L->getLatchCmpInst();
+
+  // SCEV's induction variable, restricted to a unit-step affine recurrence.
+  PHINode *Induction = L->getInductionVariable(*SE);
+  if (!Induction)
+    return nullptr;
+  // Partition bounds are integer arithmetic on the induction type, so a loop
+  // whose only induction is a pointer is out of scope.
+  if (!Induction->getType()->isIntegerTy())
+    return nullptr;
+  const SCEV *IndSCEV = SE->getSCEV(Induction);
+  // Match an affine add-recurrence and capture its constant step; accept a unit
+  // step in either direction: +1 (ascending) or -1 (descending).
+  const APInt *Step;
+  if (!match(IndSCEV, m_scev_AffineAddRec(m_SCEV(), m_scev_APInt(Step))))
+    return nullptr;
+  if (!Step->isOne() && !Step->isAllOnes())
+    return nullptr;
+  const auto *AR = cast<SCEVAddRecExpr>(IndSCEV);
+
+  // The induction's "next" value (i + 1), produced in the latch.
+  auto *StepInst = dyn_cast<Instruction>(
+      Induction->getIncomingValueForBlock(L->getLoopLatch()));
+  if (!StepInst)
+    return nullptr;
+
+  // One compare operand must be the induction, either the PHI or its step. The
+  // rebuilt latch always compares the PHI, so which operand it was is not used.
+  if (LatchCmp->getOperand(0) == Induction ||
+      LatchCmp->getOperand(0) == StepInst ||
+      LatchCmp->getOperand(1) == Induction ||
+      LatchCmp->getOperand(1) == StepInst)
+    return AR;
+  return nullptr;
+}
+
+// Decide whether the iteration ordering is signed or unsigned; returns the
+// signedness, or nullopt if it cannot be proven.
+static std::optional<bool> computeSignedness(Loop *L,
+                                             const SCEVAddRecExpr *IndAR) {
+  ICmpInst::Predicate P = L->getLatchCmpInst()->getPredicate();
+  // A relational predicate gives the ordering directly; for eq/ne fall back to
+  // the recurrence's no-wrap flags.
+  if (ICmpInst::isRelational(P))
+    return ICmpInst::isSigned(P);
+  if (IndAR->hasNoSignedWrap())
+    return true;
+  if (IndAR->hasNoUnsignedWrap())
+    return false;
+  LLVM_DEBUG(dbgs() << DEBUG_TYPE
+             ": cannot prove iteration ordering signedness\n");
+  return std::nullopt;
+}
+
+// Prove \p Pred between \p LHS and \p RHS on entry to \p L, retrying with the
+// loop guards folded in so a bound fixed by a dominating condition is seen.
+static bool isEntryGuardedByCond(ScalarEvolution &SE, Loop *L,
+                                 ICmpInst::Predicate Pred, const SCEV *LHS,
+                                 const SCEV *RHS) {
+  if (SE.isLoopEntryGuardedByCond(L, Pred, LHS, RHS))
+    return true;
+  return SE.isLoopEntryGuardedByCond(L, Pred, SE.applyLoopGuards(LHS, L),
+                                     SE.applyLoopGuards(RHS, L));
+}
+
+// Check every structural precondition and record the induction analysis.
+bool LoopSplit::isLegal() {
+  // Require a bottom-tested single-exit loop in LCSSA form. Simplify form gives
+  // the preheader, single latch and dedicated exits; the rest pin the exit to
+  // the latch, so the latch compare can be rewritten per partition.
+  if (!L->isLoopSimplifyForm() || !L->isLCSSAForm(*DT) ||
+      L->getExitingBlock() != L->getLoopLatch() || !L->getExitBlock()) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop not in expected form\n");
+    return false;
+  }
+
+  // The latch compare must exist and reside in the latch: it is rewritten in
+  // place, once per partition.
+  ICmpInst *LatchCmp = L->getLatchCmpInst();
+  if (!LatchCmp || LatchCmp->getParent() != L->getLoopLatch()) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE ": latch compare not in the loop latch\n");
+    return false;
+  }
+
+  // Exit values are unsupported. Look for an LCSSA PHI and for a use outside
+  // the loop: a token-like type cannot appear in a PHI, so LCSSA can leave a
+  // value escaping with no PHI to find.
+  if (!L->getExitBlock()->phis().empty()) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop has exit values\n");
+    return false;
+  }
+  for (BasicBlock *BB : L->blocks())
+    for (Instruction &I : *BB)
+      for (User *U : I.users())
+        if (auto *UI = dyn_cast<Instruction>(U); UI && !L->contains(UI)) {
+          LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop has exit values\n");
+          return false;
+        }
+
+  // Splitting a loop clones it, so cloning must be safe.
+  if (!L->isSafeToClone()) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop not safe to clone\n");
+    return false;
+  }
+
+  // A computable backedge-taken count fixes the iteration space we rebuild.
+  const SCEV *BTC = SE->getBackedgeTakenCount(L);
+  if (isa<SCEVCouldNotCompute>(BTC)) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop trip count uncomputable\n");
+    return false;
+  }
+
+  const SCEVAddRecExpr *IndAR = analyzeInduction(L, SE);
+  if (!IndAR) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE
+               ": no unique unit-step integer induction\n");
+    return false;
+  }
+
+  PHINode *Induction = L->getInductionVariable(*SE);
+
+  // Loop-carried values are unsupported: a later partition would have to resume
+  // the previous one's value, which needs SSA reconstruction. The induction is
+  // the exception, seeded per partition from its own start bound.
+  for (PHINode &HeaderPHI : L->getHeader()->phis())
+    if (&HeaderPHI != Induction) {
+      LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop has carried values\n");
+      return false;
+    }
+
+  std::optional<bool> Signed = computeSignedness(L, IndAR);
+  if (!Signed)
+    return false;
+  InductionIsSigned = *Signed;
+  Descending =
+      cast<SCEVConstant>(IndAR->getStepRecurrence(*SE))->getAPInt().isAllOnes();
+
+  // Start and end must share the induction type; reject any width mismatch.
+  // evaluateAtIteration coerces to the start's type for an affine recurrence,
+  // so this is defensive rather than reachable.
+  InductionEnd = IndAR->evaluateAtIteration(BTC, *SE);
+  if (InductionEnd->getType() != IndAR->getStart()->getType()) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE ": induction end/start type mismatch\n");
+    return false;
+  }
+
+  // Partition bounds and entry guards assume the space runs monotonically from
+  // start to end, so refuse one that wraps past the type extreme. A no-wrap
+  // flag on the recurrence asserts that directly.
+  bool NoWrap =
+      InductionIsSigned ? IndAR->hasNoSignedWrap() : IndAR->hasNoUnsignedWrap();
+  ICmpInst::Predicate Ordered =
+      Descending
+          ? (InductionIsSigned ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE)
+          : (InductionIsSigned ? ICmpInst::ICMP_SLE : ICmpInst::ICMP_ULE);
+  if (!NoWrap &&
+      !isEntryGuardedByCond(*SE, L, Ordered, IndAR->getStart(), InductionEnd)) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE
+               ": iteration space may wrap past the type extreme\n");
+    return false;
+  }
+
+  // A boundary can sit one step beyond the start, so that step has to be
+  // representable: from the type extreme it wraps and still compares in range.
+  // Such a loop runs one iteration anyway, which cannot be divided.
+  const SCEV *Start = IndAR->getStart();
+  if (!(Descending ? cannotBeMinInLoop(Start, L, *SE, InductionIsSigned)
+                   : cannotBeMaxInLoop(Start, L, *SE, InductionIsSigned))) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE
+               ": induction start at a type extreme, no room for a boundary\n");
+    return false;
+  }
+
+  return true;
+}
+
+//===----------------------------------------------------------------------===//
+// Transform
+//===----------------------------------------------------------------------===//
+
+// Latch "keep iterating" predicate, comparing the induction PHI against the
+// partition end: `i < sel` ascending, `i > sel` descending.
+static ICmpInst::Predicate continuePredicate(bool Signed, bool Descending) {
+  ICmpInst::Predicate P = Descending ? ICmpInst::ICMP_UGT : ICmpInst::ICMP_ULT;
+  return Signed ? ICmpInst::getSignedPredicate(P) : P;
+}
+
+// Guard "enter this partition" predicate: the latch test made non-strict, so
+// `start <= sel` ascending and `start >= sel` descending.
+static ICmpInst::Predicate guardPredicate(bool Signed, bool Descending) {
+  return ICmpInst::getNonStrictPredicate(continuePredicate(Signed, Descending));
+}
+
+static void buildEntryGuard(BasicBlock *&Preheader, BasicBlock *&EntryGuard,
+                            DominatorTree *DT, LoopInfo *LI);
+
+// Drive the whole transform: set up scratch state and run each phase in order.
+bool LoopSplit::split() {
+  PHINode *Induction = L->getInductionVariable(*SE);
+  assert(Induction && "split() requires a successful isLegal()");
+  if (getNumPartitions() < 2)
+    return false;
+
+  SplitState S;
+  // Partition 0 reuses the original loop; record its preheader/exit/guard up
+  // front.
+  PartitionInfo &P0 = Partitions[0];
+  P0.Preheader = L->getLoopPreheader();
+  P0.Exit = L->getExitBlock();
+  P0.SubLoop = L;
+  P0.IndPHI = Induction;
+  S.OuterLoop = LI->getLoopFor(P0.Exit);
+  S.Induction = Induction;
+
+  splitFinalExit(S);
+  buildEntryGuard(P0.Preheader, P0.GuardBlock, DT, LI);
+
+  // Keep the expander and its cleaner alive for the whole transform: the bounds
+  // it materializes are consumed by later phases. markResultUsed() below keeps
+  // them; without it the cleaner reclaims them.
+  SCEVExpander Expander(*SE, DEBUG_TYPE);
+  SCEVExpanderCleaner ExpanderCleaner(Expander);
+  expandPartitionBounds(S, Expander);
+  clonePartitions(S);
+  chainPartitions(S);
+  ExpanderCleaner.markResultUsed();
+
+  // The iteration space and the surrounding block structure both changed.
+  SE->forgetLoop(L);
+  SE->forgetBlockAndLoopDispositions();
+  return true;
+}
+
+// Split the final exit off the loop exit block, so the original exit can serve
+// as partition 0's dedicated exit and branch on into the guard chain.
+void LoopSplit::splitFinalExit(SplitState &S) {
+  BasicBlock *OrigExit = Partitions[0].Exit;
+
+  // Splitting at begin() moves everything into FinalExit; the exit block has no
+  // PHIs because isLegal() rejects escaping values. SplitBlock also re-parents
+  // the dominator-tree children of the exit onto FinalExit.
+  S.FinalExit = SplitBlock(OrigExit, OrigExit->begin(), DT, LI,
+                           /*MSSAU=*/nullptr, "ls.final.exit");
+}
+
+// Insert the entry guard ahead of partition 0's preheader, updating the
+// dominator tree and loop info. On return \p Preheader is the clean preheader
+// and \p EntryGuard is the new guard block dominating the chain.
+static void buildEntryGuard(BasicBlock *&Preheader, BasicBlock *&EntryGuard,
+                            DominatorTree *DT, LoopInfo *LI) {
+  // Split the preheader: the upper half becomes the guard dominating the chain,
+  // the lower half a clean preheader.
+  BasicBlock *NewPreheader =
+      SplitBlock(Preheader, Preheader->getTerminator(), DT, LI);
+  EntryGuard = Preheader;
+  Preheader = NewPreheader;
+  // Move the original preheader's name onto the new preheader, then name the
+  // guard.
+  Preheader->takeName(EntryGuard);
+  EntryGuard->setName("ls.guard0");
+}
+
+// Materialize each partition's start and clamped end in the entry guard.
+void LoopSplit::expandPartitionBounds(SplitState &S, SCEVExpander &Expander) {
+  Type *IndTy = S.Induction->getType();
+  Instruction *EntryGuardTerm = Partitions[0].GuardBlock->getTerminator();
+
+  // Expand all partition bounds in the entry guard, which dominates the whole
+  // chain (a skipped partition bypasses the original preheader).
+  const unsigned N = getNumPartitions();
+  for (unsigned I = 0; I < N; ++I) {
+    PartitionInfo &P = Partitions[I];
+
+    P.StartVal = Expander.expandCodeFor(P.StartExpr, IndTy, EntryGuardTerm);
+
+    // Clamp the end to the induction end (min ascending, max descending) so a
+    // short trip count keeps the last iteration in the right partition.
+    const SCEV *ClampedEndSCEV;
+    if (Descending)
+      ClampedEndSCEV = InductionIsSigned
+                           ? SE->getSMaxExpr(P.EndExpr, InductionEnd)
+                           : SE->getUMaxExpr(P.EndExpr, InductionEnd);
+    else
+      ClampedEndSCEV = InductionIsSigned
+                           ? SE->getSMinExpr(P.EndExpr, InductionEnd)
+                           : SE->getUMinExpr(P.EndExpr, InductionEnd);
+    P.SelEnd = Expander.expandCodeFor(ClampedEndSCEV, IndTy, EntryGuardTerm);
+  }
+}
+
+// Clone each later partition's sub-loop and create its guard and exit blocks
+// (partition 0 reuses the original loop).
+void LoopSplit::clonePartitions(SplitState &S) {
+  Function &F = *L->getHeader()->getParent();
+  LLVMContext &Ctx = F.getContext();
+
+  const unsigned N = getNumPartitions();
+  // Partition 0 reuses the original loop; clone the rest off its preheader.
+  BasicBlock *OrigPreheader = Partitions[0].Preheader;
+
+  for (unsigned I = 1; I < N; ++I) {
+    PartitionInfo &P = Partitions[I];
+    ValueToValueMapTy VMap;
+    SmallVector<BasicBlock *, 8> ClonedBlocks;
+    Loop *PL = cloneLoopWithPreheader(S.FinalExit, OrigPreheader, L, VMap,
+                                      ".ls" + Twine(I), LI, DT, ClonedBlocks);
+    remapInstructionsInBlocks(ClonedBlocks, VMap);
+    BasicBlock *PHi = PL->getLoopPreheader();
+
+    BasicBlock *Exiti =
+        BasicBlock::Create(Ctx, "ls.exit" + Twine(I), &F, S.FinalExit);
+    BasicBlock *Guardi =
+        BasicBlock::Create(Ctx, "ls.guard" + Twine(I), &F, PHi);
+    if (S.OuterLoop) {
+      S.OuterLoop->addBasicBlockToLoop(Exiti, *LI);
+      S.OuterLoop->addBasicBlockToLoop(Guardi, *LI);
+    }
+    // Placeholder terminators; both are re-pointed at the merge in pass 2.
+    UncondBrInst::Create(S.FinalExit, Exiti);
+    UncondBrInst::Create(S.FinalExit, Guardi);
+
+    // Seed the clone's induction PHI with this partition's start value.
+    auto *ClonedInduction = cast<PHINode>(VMap[S.Induction]);
+    ClonedInduction->setIncomingValueForBlock(PHi, P.StartVal);
+
+    P.GuardBlock = Guardi;
+    P.Preheader = PHi;
+    P.Exit = Exiti;
+    P.SubLoop = PL;
+    P.IndPHI = ClonedInduction;
+  }
+}
+
+// Replace a partition's latch test so it iterates only within [start, SelEnd],
+// re-point the exit edge, and carry the original branch weights over. See the
+// file comment for why the test is the strict one on the PHI.
+static void rewriteLatch(Loop *PL, PHINode *IndPHI, Value *SelEnd,
+                         BasicBlock *Exit, bool Signed, bool Descending) {
+  auto *Term = cast<CondBrInst>(PL->getLoopLatch()->getTerminator());
+  auto *Cmp = cast<ICmpInst>(Term->getCondition());
+  IRBuilder<> B(Cmp);
+  // The bound was expanded in the induction type, which is the PHI's type.
+  assert(SelEnd->getType() == IndPHI->getType() &&
+         "latch operand type mismatch");
+  ICmpInst::Predicate Pred = continuePredicate(Signed, Descending);
+  Value *NewCmp = B.CreateICmp(Pred, IndPHI, SelEnd, "itr.chk");
+  B.SetInsertPoint(Term);
+  auto *NewBr = B.CreateCondBr(NewCmp, PL->getHeader(), Exit);
+  // Carry the original latch's weights over, mapping by which successor stayed
+  // in the loop.
+  uint64_t TrueW, FalseW;
+  if (extractBranchWeights(*Term, TrueW, FalseW)) {
+    bool Succ0InLoop = PL->contains(Term->getSuccessor(0));
+    setFittedBranchWeights(
+        *NewBr, {Succ0InLoop ? TrueW : FalseW, Succ0InLoop ? FalseW : TrueW},
+        /*IsExpected=*/false);
+  }
+  Term->eraseFromParent();
+  if (Cmp->use_empty())
+    Cmp->eraseFromParent();
+}
+
+// Emit each partition's guard branch, clamp its latch, wire the partitions into
+// a chain, and update the dominator tree.
+void LoopSplit::chainPartitions(SplitState &S) {
+  const ICmpInst::Predicate GuardPred =
+      guardPredicate(InductionIsSigned, Descending);
+
+  // Emit each guard, clamp each latch, and chain partitions; a skipped
+  // partition falls through to the next guard.
+  const unsigned N = getNumPartitions();
+  IRBuilder<> B(L->getHeader()->getContext());
+
+  for (unsigned I = 0; I < N; ++I) {
+    PartitionInfo &P = Partitions[I];
+    // Where control goes when this partition is skipped or after it finishes:
+    // the next partition's guard, or the final merge for the last partition.
+    BasicBlock *MergeAfter =
+        I + 1 == N ? S.FinalExit : Partitions[I + 1].GuardBlock;
+
+    Instruction *GuardTerm = P.GuardBlock->getTerminator();
+    B.SetInsertPoint(GuardTerm);
+    // An empty partition needs no special case: its check is false and control
+    // falls through. Emitting the branch either way keeps both CFG edges, so
+    // the dominator tree stays consistent with one computed from scratch.
+    Value *Enter = B.CreateICmp(GuardPred, P.StartVal, P.SelEnd, "itr.chk");
+    auto *GuardBr = B.CreateCondBr(Enter, P.Preheader, MergeAfter);
+    // New control flow with no source profile; record the weights as unknown
+    // so profile-tracking passes are not misled.
+    setExplicitlyUnknownBranchWeightsIfProfiled(*GuardBr, DEBUG_TYPE);
+    GuardTerm->eraseFromParent();
+
+    rewriteLatch(P.SubLoop, P.IndPHI, P.SelEnd, P.Exit, InductionIsSigned,
+                 Descending);
+    P.Exit->getTerminator()->setSuccessor(0, MergeAfter);
+  }
+
+  // Patch the dominator tree directly: every partition is guarded, so a merge
+  // target is dominated by the guard that can branch straight to it.
+  for (unsigned I = 1; I < N; ++I) {
+    PartitionInfo &Cur = Partitions[I];
+    DT->addNewBlock(Cur.GuardBlock, Partitions[I - 1].GuardBlock);
+    DT->changeImmediateDominator(Cur.Preheader, Cur.GuardBlock);
+    DT->addNewBlock(Cur.Exit, Cur.SubLoop->getLoopLatch());
+  }
+  // The final exit is the last partition's merge target.
+  DT->changeImmediateDominator(S.FinalExit, Partitions.back().GuardBlock);
+}
diff --git a/llvm/lib/Transforms/Utils/LoopSplitPass.cpp b/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
new file mode 100644
index 0000000000000..3624c2526cdc2
--- /dev/null
+++ b/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
@@ -0,0 +1,143 @@
+//===- LoopSplitPass.cpp - Test driver for LoopSplit ----------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+// This pass drives LoopSplit from `opt` for testing. For every eligible loop it
+// builds partitions from the -loop-split-points offsets and splits the loop.
+// Which loops are eligible is chosen by -loop-split-depth; the default is the
+// innermost ones.
+//
+//===----------------------------------------------------------------------===//
+
+#include "llvm/Transforms/Utils/LoopSplitPass.h"
+#include "llvm/ADT/STLExtras.h"
+#include "llvm/ADT/SmallVector.h"
+#include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/ScalarEvolution.h"
+#include "llvm/Analysis/ScalarEvolutionExpressions.h"
+#include "llvm/Analysis/ScalarEvolutionPatternMatch.h"
+#include "llvm/IR/Dominators.h"
+#include "llvm/IR/Function.h"
+#include "llvm/Support/CommandLine.h"
+#include "llvm/Support/Debug.h"
+#include "llvm/Transforms/Utils/LoopSplit.h"
+
+using namespace llvm;
+using namespace llvm::SCEVPatternMatch;
+
+#define DEBUG_TYPE "loop-split"
+
+static cl::list<unsigned>
+    SplitPoints("loop-split-points",
+                cl::desc("Iteration offsets (relative to the induction start) "
+                         "at which to split each loop"),
+                cl::CommaSeparated);
+
+static cl::opt<unsigned> SplitDepth(
+    "loop-split-depth",
+    cl::desc(
+        "Split the loops at this nesting depth (1 is outermost) instead of "
+        "the innermost ones"),
+    cl::init(0));
+
+// Build the partition list for \p L from the command-line split offsets and run
+// the transform. Returns true if the loop was split.
+static bool splitLoop(Loop *L, ScalarEvolution &SE, DominatorTree &DT,
+                      LoopInfo &LI) {
+  LoopSplit LS(L, &LI, &SE, &DT);
+  if (!LS.isLegal()) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop is not legal for splitting\n");
+    return false;
+  }
+
+  // isLegal() has already established this shape.
+  const SCEV *IndVarSCEV = SE.getSCEV(LS.getInductionVariable());
+  const SCEV *Start;
+  const APInt *StepC;
+  [[maybe_unused]] bool Matched = match(
+      IndVarSCEV, m_scev_AffineAddRec(m_SCEV(Start), m_scev_APInt(StepC)));
+  assert(Matched && "isLegal() guarantees a unit-step affine induction");
+
+  const SCEV *BTC = SE.getBackedgeTakenCount(L);
+  const SCEV *End = LS.getInductionEnd();
+  Type *Ty = Start->getType();
+  unsigned BitWidth = Ty->getIntegerBitWidth();
+  // The backedge-taken count is a separate expression and need not share the
+  // induction's width, so coerce it before doing arithmetic in that type.
+  const SCEV *Count = SE.getTruncateOrZeroExtend(BTC, Ty);
+
+  // Build boundaries in iteration order, stepping away from Start by each
+  // offset (down for a descending loop). Each offset opens a new partition at
+  // iteration `Start +/- offset`; the previous partition ends one step before.
+  bool Descending = StepC->isAllOnes();
+
+  // Boundaries must be increasing and distinct to tile the space, so sort and
+  // unique the offsets. Drop any that do not fit the induction type; truncating
+  // would reorder them and the partitions would overlap.
+  SmallVector<unsigned, 4> Offsets;
+  for (unsigned Offset : SplitPoints)
+    if (BitWidth >= 32 || Offset < (1u << BitWidth))
+      Offsets.push_back(Offset);
+  llvm::sort(Offsets);
+  Offsets.erase(llvm::unique(Offsets), Offsets.end());
+
+  const SCEV *PrevStart = Start;
+  const SCEV *One = SE.getOne(Ty);
+  for (unsigned Offset : Offsets) {
+    // Clamp into [1, BTC] so each boundary stays in the space; Start +/- BTC is
+    // the last iteration. The umax reaches one past it when BTC is zero, which
+    // isLegal() proved representable.
+    const SCEV *Off = SE.getConstant(Ty, Offset);
+    Off = SE.getUMaxExpr(One, SE.getUMinExpr(Off, Count));
+    const SCEV *Point =
+        Descending ? SE.getMinusSCEV(Start, Off) : SE.getAddExpr(Start, Off);
+    const SCEV *PrevEnd =
+        Descending ? SE.getAddExpr(Point, One) : SE.getMinusSCEV(Point, One);
+    LS.addPartition(PrevStart, PrevEnd);
+    PrevStart = Point;
+  }
+  // The final partition runs to the iteration-space end.
+  LS.addPartition(PrevStart, End);
+
+  if (LS.getNumPartitions() < 2)
+    return false;
+
+  return LS.split();
+}
+
+// Split the selected loops in \p F at the command-line offsets.
+PreservedAnalyses LoopSplitPass::run(Function &F, FunctionAnalysisManager &AM) {
+  if (SplitPoints.empty())
+    return PreservedAnalyses::all();
+
+  auto &LI = AM.getResult<LoopAnalysis>(F);
+  auto &SE = AM.getResult<ScalarEvolutionAnalysis>(F);
+  auto &DT = AM.getResult<DominatorTreeAnalysis>(F);
+
+  // Collect the loops up front: the transform creates new sibling loops that we
+  // must not revisit. Depth 0 means no depth was given, so take the innermost
+  // loops; LoopSplit itself works at any depth.
+  SmallVector<Loop *, 4> Worklist;
+  for (Loop *L : LI.getLoopsInPreorder())
+    if (SplitDepth ? L->getLoopDepth() == SplitDepth : L->isInnermost())
+      Worklist.push_back(L);
+
+  bool Changed = false;
+  for (Loop *L : Worklist)
+    Changed |= splitLoop(L, SE, DT, LI);
+
+  if (!Changed)
+    return PreservedAnalyses::all();
+
+  // LoopSplit patches the dominator tree and loop info as it goes, so keeping
+  // them lets `verify<domtree>` and `verify<loops>` check those updates instead
+  // of a freshly recomputed copy.
+  PreservedAnalyses PA;
+  PA.preserve<DominatorTreeAnalysis>();
+  PA.preserve<LoopAnalysis>();
+  return PA;
+}
diff --git a/llvm/test/Transforms/LoopSplit/basic.ll b/llvm/test/Transforms/LoopSplit/basic.ll
new file mode 100644
index 0000000000000..df513a5a37286
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/basic.ll
@@ -0,0 +1,263 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=50 -S < %s | FileCheck %s
+
+; A counted loop is rewritten into a guarded chain of two sub-loops covering
+; [0, 49] and [50, end]. Every bound is materialized in ls.guard0, and each
+; latch is clamped to its partition's end.
+
+define void @split_signed(ptr %a, i64 %n) {
+; CHECK-LABEL: define void @split_signed(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 50)
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[IV]], [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK2]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i64 [ [[UMAX]], %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; CHECK-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[IV_LS1]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nsw i64 %iv, 1
+  %c = icmp slt i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; An unsigned latch predicate makes the iteration ordering unsigned, so the end
+; clamp uses umin and the guard/latch predicates are unsigned.
+
+define void @split_unsigned(ptr %a, i64 %n) {
+; CHECK-LABEL: define void @split_unsigned(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[UMAX]], -1
+; CHECK-NEXT:    [[UMIN1:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 50)
+; CHECK-NEXT:    [[UMAX1:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN1]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX1]], -1
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp ule i64 0, [[UMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw i64 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp ult i64 [[IV]], [[UMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK2:%.*]] = icmp ule i64 [[UMAX1]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK2]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i64 [ [[UMAX1]], %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; CHECK-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add nuw i64 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp ult i64 [[IV_LS1]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nuw i64 %iv, 1
+  %c = icmp ult i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; When the latch compares the induction PHI rather than its step value, the
+; rebuilt latch stays strict (slt) instead of becoming inclusive (sle).
+
+define void @latch_compares_phi(ptr %a, i64 %m) {
+; CHECK-LABEL: define void @latch_compares_phi(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[M]], i64 0)
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[SMAX]], i64 50)
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[UMAX]], -1
+; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[SMAX]], i64 [[TMP0]])
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[IV]], [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[SMAX]]
+; CHECK-NEXT:    br i1 [[ITR_CHK2]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i64 [ [[UMAX]], %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; CHECK-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[IV_LS1]], [[SMAX]]
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nsw i64 %iv, 1
+  %c = icmp slt i64 %iv, %m
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The inner loop of a nest is the innermost loop, so it is the one split. The
+; new guard, preheader and exit blocks must join the surrounding loop.
+
+define void @nested_inner(ptr %a, i64 %n, i64 %m) {
+; CHECK-LABEL: define void @nested_inner(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[M]], i64 1)
+; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 50)
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT:    br label %[[LS_GUARD0:.*]]
+; CHECK:       [[LS_GUARD0]]:
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[INNER_PH:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[INNER_PH]]:
+; CHECK-NEXT:    br label %[[INNER_HEADER:.*]]
+; CHECK:       [[INNER_HEADER]]:
+; CHECK-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[INNER_PH]] ], [ [[J_NEXT:%.*]], %[[INNER_HEADER]] ]
+; CHECK-NEXT:    [[IDX:%.*]] = add i64 [[I]], [[J]]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX]]
+; CHECK-NEXT:    store i64 [[J]], ptr [[P]], align 4
+; CHECK-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[J]], [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[INNER_HEADER]], label %[[INNER_EXIT:.*]]
+; CHECK:       [[INNER_EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK2]], label %[[INNER_PH_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[INNER_PH_LS1]]:
+; CHECK-NEXT:    br label %[[INNER_HEADER_LS1:.*]]
+; CHECK:       [[INNER_HEADER_LS1]]:
+; CHECK-NEXT:    [[J_LS1:%.*]] = phi i64 [ [[UMAX]], %[[INNER_PH_LS1]] ], [ [[J_NEXT_LS1:%.*]], %[[INNER_HEADER_LS1]] ]
+; CHECK-NEXT:    [[IDX_LS1:%.*]] = add i64 [[I]], [[J_LS1]]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX_LS1]]
+; CHECK-NEXT:    store i64 [[J_LS1]], ptr [[P_LS1]], align 4
+; CHECK-NEXT:    [[J_NEXT_LS1]] = add nsw i64 [[J_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[J_LS1]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[INNER_HEADER_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    br label %[[OUTER_LATCH]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT:    [[OC:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[OC]], label %[[OUTER_HEADER]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %outer.latch ]
+  br label %inner.ph
+
+inner.ph:
+  br label %inner.header
+
+inner.header:
+  %j = phi i64 [ 0, %inner.ph ], [ %j.next, %inner.header ]
+  %idx = add i64 %i, %j
+  %p = getelementptr inbounds i64, ptr %a, i64 %idx
+  store i64 %j, ptr %p
+  %j.next = add nsw i64 %j, 1
+  %ic = icmp slt i64 %j.next, %m
+  br i1 %ic, label %inner.header, label %inner.exit
+
+inner.exit:
+  br label %outer.latch
+
+outer.latch:
+  %i.next = add nsw i64 %i, 1
+  %oc = icmp slt i64 %i.next, %n
+  br i1 %oc, label %outer.header, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopSplit/branch-weights.ll b/llvm/test/Transforms/LoopSplit/branch-weights.ll
new file mode 100644
index 0000000000000..4bdd4e37f811d
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/branch-weights.ll
@@ -0,0 +1,70 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=50 -S < %s | FileCheck %s
+
+; The original latch carries branch weights; splitting must propagate them onto
+; each partition's clamped latch. The newly created partition guards have no
+; source profile, so in a profiled function they are marked with unknown
+; weights.
+
+define void @basic(ptr %a, i64 %n) !prof !0 {
+; CHECK-LABEL: define void @basic(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) !prof [[PROF0:![0-9]+]] {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 50)
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]], !prof [[PROF1:![0-9]+]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[IV]], [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP]], label %[[EXIT:.*]], !prof [[PROF2:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK2]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]], !prof [[PROF1]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i64 [ [[UMAX]], %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr i64, ptr [[A]], i64 [[IV_LS1]]
+; CHECK-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add i64 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[IV_LS1]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]], !prof [[PROF2]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add i64 %iv, 1
+  %ec = icmp slt i64 %iv.next, %n
+  br i1 %ec, label %loop, label %exit, !prof !1
+
+exit:
+  ret void
+}
+
+!0 = !{!"function_entry_count", i64 1000}
+!1 = !{!"branch_weights", i32 100, i32 1}
+;.
+; CHECK: [[PROF0]] = !{!"function_entry_count", i64 1000}
+; CHECK: [[PROF1]] = !{!"unknown", !"loop-split"}
+; CHECK: [[PROF2]] = !{!"branch_weights", i32 100, i32 1}
+;.
diff --git a/llvm/test/Transforms/LoopSplit/constant-trip-count.ll b/llvm/test/Transforms/LoopSplit/constant-trip-count.ll
new file mode 100644
index 0000000000000..9cdabd0056539
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/constant-trip-count.ll
@@ -0,0 +1,153 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=50 -S < %s | FileCheck %s --check-prefix=INSIDE
+; RUN: opt -passes=loop-split -loop-split-points=200 -S < %s | FileCheck %s --check-prefix=BEYOND
+
+; With a constant trip count every bound folds, so no smin/smax calls survive.
+;
+; For a split point inside the space the two partitions are [0, 49] and
+; [50, 99], and both guards fold to true.
+;
+; For a split point past the end, the clamp keeps the last iteration in
+; partition 0 and partition 1's guard folds to false so it never runs. Both
+; edges stay in the CFG, which keeps the dominator tree consistent with one
+; computed from scratch; later passes delete the dead partition.
+
+define void @tc100(ptr %a) {
+; IN-LABEL: define void @tc100(
+; IN-SAME: ptr [[A:%.*]]) {
+; IN-NEXT:  [[LS_GUARD0:.*:]]
+; IN-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; IN:       [[ENTRY]]:
+; IN-NEXT:    br label %[[LOOP:.*]]
+; IN:       [[LOOP]]:
+; IN-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; IN-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; IN-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; IN-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; IN-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 [[IV_NEXT]], 49
+; IN-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
+; IN:       [[EXIT]]:
+; IN-NEXT:    br label %[[LS_GUARD1]]
+; IN:       [[LS_GUARD1]]:
+; IN-NEXT:    br i1 true, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; IN:       [[ENTRY_LS1]]:
+; IN-NEXT:    br label %[[LOOP_LS1:.*]]
+; IN:       [[LOOP_LS1]]:
+; IN-NEXT:    [[IV_LS1:%.*]] = phi i64 [ 50, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; IN-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; IN-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; IN-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], 1
+; IN-NEXT:    [[ITR_CHK1:%.*]] = icmp sle i64 [[IV_NEXT_LS1]], 99
+; IN-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; IN:       [[LS_EXIT1]]:
+; IN-NEXT:    br label %[[LS_FINAL_EXIT]]
+; IN:       [[LS_FINAL_EXIT]]:
+; IN-NEXT:    ret void
+;
+; OUT-LABEL: define void @tc100(
+; OUT-SAME: ptr [[A:%.*]]) {
+; OUT-NEXT:  [[LS_GUARD0:.*:]]
+; OUT-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; OUT:       [[ENTRY]]:
+; OUT-NEXT:    br label %[[LOOP:.*]]
+; OUT:       [[LOOP]]:
+; OUT-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; OUT-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; OUT-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; OUT-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; OUT-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 [[IV_NEXT]], 99
+; OUT-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
+; OUT:       [[EXIT]]:
+; OUT-NEXT:    br label %[[LS_GUARD1]]
+; OUT:       [[LS_GUARD1]]:
+; OUT-NEXT:    br i1 false, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; OUT:       [[ENTRY_LS1]]:
+; OUT-NEXT:    br label %[[LOOP_LS1:.*]]
+; OUT:       [[LOOP_LS1]]:
+; OUT-NEXT:    [[IV_LS1:%.*]] = phi i64 [ 200, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; OUT-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; OUT-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; OUT-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], 1
+; OUT-NEXT:    [[ITR_CHK1:%.*]] = icmp sle i64 [[IV_NEXT_LS1]], 99
+; OUT-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; OUT:       [[LS_EXIT1]]:
+; OUT-NEXT:    br label %[[LS_FINAL_EXIT]]
+; OUT:       [[LS_FINAL_EXIT]]:
+; OUT-NEXT:    ret void
+;
+; INSIDE-LABEL: define void @tc100(
+; INSIDE-SAME: ptr [[A:%.*]]) {
+; INSIDE-NEXT:  [[LS_GUARD0:.*:]]
+; INSIDE-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; INSIDE:       [[ENTRY]]:
+; INSIDE-NEXT:    br label %[[LOOP:.*]]
+; INSIDE:       [[LOOP]]:
+; INSIDE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; INSIDE-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; INSIDE-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; INSIDE-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; INSIDE-NEXT:    [[ITR_CHK:%.*]] = icmp slt i64 [[IV]], 49
+; INSIDE-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
+; INSIDE:       [[EXIT]]:
+; INSIDE-NEXT:    br label %[[LS_GUARD1]]
+; INSIDE:       [[LS_GUARD1]]:
+; INSIDE-NEXT:    br i1 true, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; INSIDE:       [[ENTRY_LS1]]:
+; INSIDE-NEXT:    br label %[[LOOP_LS1:.*]]
+; INSIDE:       [[LOOP_LS1]]:
+; INSIDE-NEXT:    [[IV_LS1:%.*]] = phi i64 [ 50, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; INSIDE-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; INSIDE-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; INSIDE-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], 1
+; INSIDE-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[IV_LS1]], 99
+; INSIDE-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; INSIDE:       [[LS_EXIT1]]:
+; INSIDE-NEXT:    br label %[[LS_FINAL_EXIT]]
+; INSIDE:       [[LS_FINAL_EXIT]]:
+; INSIDE-NEXT:    ret void
+;
+; BEYOND-LABEL: define void @tc100(
+; BEYOND-SAME: ptr [[A:%.*]]) {
+; BEYOND-NEXT:  [[LS_GUARD0:.*:]]
+; BEYOND-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; BEYOND:       [[ENTRY]]:
+; BEYOND-NEXT:    br label %[[LOOP:.*]]
+; BEYOND:       [[LOOP]]:
+; BEYOND-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; BEYOND-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; BEYOND-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; BEYOND-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; BEYOND-NEXT:    [[ITR_CHK:%.*]] = icmp slt i64 [[IV]], 98
+; BEYOND-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
+; BEYOND:       [[EXIT]]:
+; BEYOND-NEXT:    br label %[[LS_GUARD1]]
+; BEYOND:       [[LS_GUARD1]]:
+; BEYOND-NEXT:    br i1 true, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; BEYOND:       [[ENTRY_LS1]]:
+; BEYOND-NEXT:    br label %[[LOOP_LS1:.*]]
+; BEYOND:       [[LOOP_LS1]]:
+; BEYOND-NEXT:    [[IV_LS1:%.*]] = phi i64 [ 99, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; BEYOND-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; BEYOND-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; BEYOND-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], 1
+; BEYOND-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[IV_LS1]], 99
+; BEYOND-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; BEYOND:       [[LS_EXIT1]]:
+; BEYOND-NEXT:    br label %[[LS_FINAL_EXIT]]
+; BEYOND:       [[LS_FINAL_EXIT]]:
+; BEYOND-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nsw i64 %iv, 1
+  %c = icmp slt i64 %iv.next, 100
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopSplit/descending.ll b/llvm/test/Transforms/LoopSplit/descending.ll
new file mode 100644
index 0000000000000..9d3b047a07bc5
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/descending.ll
@@ -0,0 +1,170 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=20 -S < %s | FileCheck %s
+
+; A step of -1 mirrors the structure: partitions run high to low, the end clamp
+; becomes a max, and the guard and latch predicates flip to >= and >.
+
+define void @descending_signed(ptr %a, i64 %n) {
+; CHECK-LABEL: define void @descending_signed(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[N]], i64 99)
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i64 99, [[SMIN]]
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP2]], i64 20)
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = sub i64 101, [[UMAX]]
+; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMIN]], 1
+; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[TMP1]], i64 [[TMP0]])
+; CHECK-NEXT:    [[TMP3:%.*]] = sub i64 100, [[UMAX]]
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sge i64 100, [[SMAX]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 100, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], -1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp sgt i64 [[IV]], [[SMAX]]
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK2:%.*]] = icmp sge i64 [[TMP3]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK2]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i64 [ [[TMP3]], %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; CHECK-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], -1
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp sgt i64 [[IV_LS1]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 100, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nsw i64 %iv, -1
+  %c = icmp sgt i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The same shape with an unsigned latch predicate and constant bounds, so the
+; ordering is unsigned and the clamp uses umax.
+
+define void @descending_unsigned(ptr %a) {
+; CHECK-LABEL: define void @descending_unsigned(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 100, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], -1
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp ugt i64 [[IV]], 81
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    br i1 true, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i64 [ 80, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; CHECK-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add i64 [[IV_LS1]], -1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp ugt i64 [[IV_LS1]], 11
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 100, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add i64 %iv, -1
+  %c = icmp ugt i64 %iv.next, 10
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; A descending loop whose latch compares the PHI keeps a strict predicate.
+
+define void @descending_latch_compares_phi(ptr %a, i64 %m) {
+; CHECK-LABEL: define void @descending_latch_compares_phi(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[M:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[M]], i64 100)
+; CHECK-NEXT:    [[TMP0:%.*]] = sub i64 100, [[SMIN]]
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 20)
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = sub i64 101, [[UMAX]]
+; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[SMIN]], i64 [[TMP1]])
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i64 100, [[UMAX]]
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sge i64 100, [[SMAX]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 100, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], -1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp sgt i64 [[IV]], [[SMAX]]
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK2:%.*]] = icmp sge i64 [[TMP2]], [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK2]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i64 [ [[TMP2]], %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; CHECK-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], -1
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp sgt i64 [[IV_LS1]], [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 100, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nsw i64 %iv, -1
+  %c = icmp sgt i64 %iv, %m
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopSplit/loop-depth.ll b/llvm/test/Transforms/LoopSplit/loop-depth.ll
new file mode 100644
index 0000000000000..5b4d0680eec8f
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/loop-depth.ll
@@ -0,0 +1,638 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=2 -S < %s | FileCheck %s --check-prefix=INNERMOST
+; RUN: opt -passes=loop-split -loop-split-points=2 -loop-split-depth=1 -S < %s | FileCheck %s --check-prefix=DEPTH1
+; RUN: opt -passes=loop-split -loop-split-points=2 -loop-split-depth=2 -S < %s | FileCheck %s --check-prefix=DEPTH2
+
+; LoopSplit works on a loop at any nesting depth; -loop-split-depth selects which
+; one, and without it the innermost loops are taken. Splitting an outer loop
+; clones its sub-loops with it.
+
+; Two deep. INNERMOST and DEPTH2 both pick the inner loop; DEPTH1 picks the
+; outer one and clones the inner loop into each partition.
+
+define void @nest2(ptr %a, i64 %n, i64 %m) {
+; INNERMOST-LABEL: define void @nest2(
+; INNERMOST-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; INNERMOST-NEXT:  [[ENTRY:.*]]:
+; INNERMOST-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[M]], i64 1)
+; INNERMOST-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; INNERMOST-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
+; INNERMOST-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; INNERMOST-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; INNERMOST-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; INNERMOST-NEXT:    br label %[[OUTER_HEADER:.*]]
+; INNERMOST:       [[OUTER_HEADER]]:
+; INNERMOST-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; INNERMOST-NEXT:    br label %[[LS_GUARD0:.*]]
+; INNERMOST:       [[LS_GUARD0]]:
+; INNERMOST-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; INNERMOST-NEXT:    br i1 [[ITR_CHK]], label %[[INNER_PH:.*]], label %[[LS_GUARD1:.*]]
+; INNERMOST:       [[INNER_PH]]:
+; INNERMOST-NEXT:    br label %[[INNER_HEADER:.*]]
+; INNERMOST:       [[INNER_HEADER]]:
+; INNERMOST-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[INNER_PH]] ], [ [[J_NEXT:%.*]], %[[INNER_HEADER]] ]
+; INNERMOST-NEXT:    [[IDX:%.*]] = add i64 [[I]], [[J]]
+; INNERMOST-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX]]
+; INNERMOST-NEXT:    store i64 [[IDX]], ptr [[P]], align 4
+; INNERMOST-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; INNERMOST-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[J]], [[SMIN]]
+; INNERMOST-NEXT:    br i1 [[ITR_CHK1]], label %[[INNER_HEADER]], label %[[INNER_EXIT:.*]]
+; INNERMOST:       [[INNER_EXIT]]:
+; INNERMOST-NEXT:    br label %[[LS_GUARD1]]
+; INNERMOST:       [[LS_GUARD1]]:
+; INNERMOST-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; INNERMOST-NEXT:    br i1 [[ITR_CHK2]], label %[[INNER_PH_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; INNERMOST:       [[INNER_PH_LS1]]:
+; INNERMOST-NEXT:    br label %[[INNER_HEADER_LS1:.*]]
+; INNERMOST:       [[INNER_HEADER_LS1]]:
+; INNERMOST-NEXT:    [[J_LS1:%.*]] = phi i64 [ [[UMAX]], %[[INNER_PH_LS1]] ], [ [[J_NEXT_LS1:%.*]], %[[INNER_HEADER_LS1]] ]
+; INNERMOST-NEXT:    [[IDX_LS1:%.*]] = add i64 [[I]], [[J_LS1]]
+; INNERMOST-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX_LS1]]
+; INNERMOST-NEXT:    store i64 [[IDX_LS1]], ptr [[P_LS1]], align 4
+; INNERMOST-NEXT:    [[J_NEXT_LS1]] = add nsw i64 [[J_LS1]], 1
+; INNERMOST-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[J_LS1]], [[TMP0]]
+; INNERMOST-NEXT:    br i1 [[ITR_CHK3]], label %[[INNER_HEADER_LS1]], label %[[LS_EXIT1:.*]]
+; INNERMOST:       [[LS_EXIT1]]:
+; INNERMOST-NEXT:    br label %[[LS_FINAL_EXIT]]
+; INNERMOST:       [[LS_FINAL_EXIT]]:
+; INNERMOST-NEXT:    br label %[[OUTER_LATCH]]
+; INNERMOST:       [[OUTER_LATCH]]:
+; INNERMOST-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; INNERMOST-NEXT:    [[OC:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; INNERMOST-NEXT:    br i1 [[OC]], label %[[OUTER_HEADER]], label %[[EXIT:.*]]
+; INNERMOST:       [[EXIT]]:
+; INNERMOST-NEXT:    ret void
+;
+; DEPTH1-LABEL: define void @nest2(
+; DEPTH1-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; DEPTH1-NEXT:  [[LS_GUARD0:.*:]]
+; DEPTH1-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; DEPTH1-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; DEPTH1-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
+; DEPTH1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; DEPTH1-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; DEPTH1-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; DEPTH1-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; DEPTH1:       [[ENTRY]]:
+; DEPTH1-NEXT:    br label %[[OUTER_HEADER:.*]]
+; DEPTH1:       [[OUTER_HEADER]]:
+; DEPTH1-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; DEPTH1-NEXT:    br label %[[INNER_PH:.*]]
+; DEPTH1:       [[INNER_PH]]:
+; DEPTH1-NEXT:    br label %[[INNER_HEADER:.*]]
+; DEPTH1:       [[INNER_HEADER]]:
+; DEPTH1-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[INNER_PH]] ], [ [[J_NEXT:%.*]], %[[INNER_HEADER]] ]
+; DEPTH1-NEXT:    [[IDX:%.*]] = add i64 [[I]], [[J]]
+; DEPTH1-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX]]
+; DEPTH1-NEXT:    store i64 [[IDX]], ptr [[P]], align 4
+; DEPTH1-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; DEPTH1-NEXT:    [[IC:%.*]] = icmp slt i64 [[J_NEXT]], [[M]]
+; DEPTH1-NEXT:    br i1 [[IC]], label %[[INNER_HEADER]], label %[[INNER_EXIT:.*]]
+; DEPTH1:       [[INNER_EXIT]]:
+; DEPTH1-NEXT:    br label %[[OUTER_LATCH]]
+; DEPTH1:       [[OUTER_LATCH]]:
+; DEPTH1-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; DEPTH1-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[I]], [[SMIN]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK1]], label %[[OUTER_HEADER]], label %[[EXIT:.*]]
+; DEPTH1:       [[EXIT]]:
+; DEPTH1-NEXT:    br label %[[LS_GUARD1]]
+; DEPTH1:       [[LS_GUARD1]]:
+; DEPTH1-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK2]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; DEPTH1:       [[ENTRY_LS1]]:
+; DEPTH1-NEXT:    br label %[[OUTER_HEADER_LS1:.*]]
+; DEPTH1:       [[OUTER_HEADER_LS1]]:
+; DEPTH1-NEXT:    [[I_LS1:%.*]] = phi i64 [ [[UMAX]], %[[ENTRY_LS1]] ], [ [[I_NEXT_LS1:%.*]], %[[OUTER_LATCH_LS1:.*]] ]
+; DEPTH1-NEXT:    br label %[[INNER_PH_LS1:.*]]
+; DEPTH1:       [[INNER_PH_LS1]]:
+; DEPTH1-NEXT:    br label %[[INNER_HEADER_LS1:.*]]
+; DEPTH1:       [[INNER_HEADER_LS1]]:
+; DEPTH1-NEXT:    [[J_LS1:%.*]] = phi i64 [ 0, %[[INNER_PH_LS1]] ], [ [[J_NEXT_LS1:%.*]], %[[INNER_HEADER_LS1]] ]
+; DEPTH1-NEXT:    [[IDX_LS1:%.*]] = add i64 [[I_LS1]], [[J_LS1]]
+; DEPTH1-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX_LS1]]
+; DEPTH1-NEXT:    store i64 [[IDX_LS1]], ptr [[P_LS1]], align 4
+; DEPTH1-NEXT:    [[J_NEXT_LS1]] = add nsw i64 [[J_LS1]], 1
+; DEPTH1-NEXT:    [[IC_LS1:%.*]] = icmp slt i64 [[J_NEXT_LS1]], [[M]]
+; DEPTH1-NEXT:    br i1 [[IC_LS1]], label %[[INNER_HEADER_LS1]], label %[[INNER_EXIT_LS1:.*]]
+; DEPTH1:       [[INNER_EXIT_LS1]]:
+; DEPTH1-NEXT:    br label %[[OUTER_LATCH_LS1]]
+; DEPTH1:       [[OUTER_LATCH_LS1]]:
+; DEPTH1-NEXT:    [[I_NEXT_LS1]] = add nsw i64 [[I_LS1]], 1
+; DEPTH1-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[I_LS1]], [[TMP0]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK3]], label %[[OUTER_HEADER_LS1]], label %[[LS_EXIT1:.*]]
+; DEPTH1:       [[LS_EXIT1]]:
+; DEPTH1-NEXT:    br label %[[LS_FINAL_EXIT]]
+; DEPTH1:       [[LS_FINAL_EXIT]]:
+; DEPTH1-NEXT:    ret void
+;
+; DEPTH2-LABEL: define void @nest2(
+; DEPTH2-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; DEPTH2-NEXT:  [[ENTRY:.*]]:
+; DEPTH2-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[M]], i64 1)
+; DEPTH2-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; DEPTH2-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
+; DEPTH2-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; DEPTH2-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; DEPTH2-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; DEPTH2-NEXT:    br label %[[OUTER_HEADER:.*]]
+; DEPTH2:       [[OUTER_HEADER]]:
+; DEPTH2-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; DEPTH2-NEXT:    br label %[[LS_GUARD0:.*]]
+; DEPTH2:       [[LS_GUARD0]]:
+; DEPTH2-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; DEPTH2-NEXT:    br i1 [[ITR_CHK]], label %[[INNER_PH:.*]], label %[[LS_GUARD1:.*]]
+; DEPTH2:       [[INNER_PH]]:
+; DEPTH2-NEXT:    br label %[[INNER_HEADER:.*]]
+; DEPTH2:       [[INNER_HEADER]]:
+; DEPTH2-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[INNER_PH]] ], [ [[J_NEXT:%.*]], %[[INNER_HEADER]] ]
+; DEPTH2-NEXT:    [[IDX:%.*]] = add i64 [[I]], [[J]]
+; DEPTH2-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX]]
+; DEPTH2-NEXT:    store i64 [[IDX]], ptr [[P]], align 4
+; DEPTH2-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; DEPTH2-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[J]], [[SMIN]]
+; DEPTH2-NEXT:    br i1 [[ITR_CHK1]], label %[[INNER_HEADER]], label %[[INNER_EXIT:.*]]
+; DEPTH2:       [[INNER_EXIT]]:
+; DEPTH2-NEXT:    br label %[[LS_GUARD1]]
+; DEPTH2:       [[LS_GUARD1]]:
+; DEPTH2-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; DEPTH2-NEXT:    br i1 [[ITR_CHK2]], label %[[INNER_PH_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; DEPTH2:       [[INNER_PH_LS1]]:
+; DEPTH2-NEXT:    br label %[[INNER_HEADER_LS1:.*]]
+; DEPTH2:       [[INNER_HEADER_LS1]]:
+; DEPTH2-NEXT:    [[J_LS1:%.*]] = phi i64 [ [[UMAX]], %[[INNER_PH_LS1]] ], [ [[J_NEXT_LS1:%.*]], %[[INNER_HEADER_LS1]] ]
+; DEPTH2-NEXT:    [[IDX_LS1:%.*]] = add i64 [[I]], [[J_LS1]]
+; DEPTH2-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX_LS1]]
+; DEPTH2-NEXT:    store i64 [[IDX_LS1]], ptr [[P_LS1]], align 4
+; DEPTH2-NEXT:    [[J_NEXT_LS1]] = add nsw i64 [[J_LS1]], 1
+; DEPTH2-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[J_LS1]], [[TMP0]]
+; DEPTH2-NEXT:    br i1 [[ITR_CHK3]], label %[[INNER_HEADER_LS1]], label %[[LS_EXIT1:.*]]
+; DEPTH2:       [[LS_EXIT1]]:
+; DEPTH2-NEXT:    br label %[[LS_FINAL_EXIT]]
+; DEPTH2:       [[LS_FINAL_EXIT]]:
+; DEPTH2-NEXT:    br label %[[OUTER_LATCH]]
+; DEPTH2:       [[OUTER_LATCH]]:
+; DEPTH2-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; DEPTH2-NEXT:    [[OC:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; DEPTH2-NEXT:    br i1 [[OC]], label %[[OUTER_HEADER]], label %[[EXIT:.*]]
+; DEPTH2:       [[EXIT]]:
+; DEPTH2-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %outer.latch ]
+  br label %inner.ph
+
+inner.ph:
+  br label %inner.header
+
+inner.header:
+  %j = phi i64 [ 0, %inner.ph ], [ %j.next, %inner.header ]
+  %idx = add i64 %i, %j
+  %p = getelementptr inbounds i64, ptr %a, i64 %idx
+  store i64 %idx, ptr %p
+  %j.next = add nsw i64 %j, 1
+  %ic = icmp slt i64 %j.next, %m
+  br i1 %ic, label %inner.header, label %inner.exit
+
+inner.exit:
+  br label %outer.latch
+
+outer.latch:
+  %i.next = add nsw i64 %i, 1
+  %oc = icmp slt i64 %i.next, %n
+  br i1 %oc, label %outer.header, label %exit
+
+exit:
+  ret void
+}
+
+; Three deep, so DEPTH2 selects the middle loop: one that is neither outermost
+; nor innermost.
+
+define void @nest3(ptr %a, i64 %n) {
+; INNERMOST-LABEL: define void @nest3(
+; INNERMOST-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; INNERMOST-NEXT:  [[ENTRY:.*]]:
+; INNERMOST-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; INNERMOST-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; INNERMOST-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
+; INNERMOST-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; INNERMOST-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; INNERMOST-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; INNERMOST-NEXT:    br label %[[O_HEADER:.*]]
+; INNERMOST:       [[O_HEADER]]:
+; INNERMOST-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[O_LATCH:.*]] ]
+; INNERMOST-NEXT:    br label %[[M_PH:.*]]
+; INNERMOST:       [[M_PH]]:
+; INNERMOST-NEXT:    br label %[[M_HEADER:.*]]
+; INNERMOST:       [[M_HEADER]]:
+; INNERMOST-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[M_PH]] ], [ [[J_NEXT:%.*]], %[[M_LATCH:.*]] ]
+; INNERMOST-NEXT:    br label %[[LS_GUARD0:.*]]
+; INNERMOST:       [[LS_GUARD0]]:
+; INNERMOST-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; INNERMOST-NEXT:    br i1 [[ITR_CHK]], label %[[I_PH:.*]], label %[[LS_GUARD1:.*]]
+; INNERMOST:       [[I_PH]]:
+; INNERMOST-NEXT:    br label %[[I_HEADER:.*]]
+; INNERMOST:       [[I_HEADER]]:
+; INNERMOST-NEXT:    [[K:%.*]] = phi i64 [ 0, %[[I_PH]] ], [ [[K_NEXT:%.*]], %[[I_HEADER]] ]
+; INNERMOST-NEXT:    [[IDX:%.*]] = add i64 [[J]], [[K]]
+; INNERMOST-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX]]
+; INNERMOST-NEXT:    store i64 [[IDX]], ptr [[P]], align 4
+; INNERMOST-NEXT:    [[K_NEXT]] = add nsw i64 [[K]], 1
+; INNERMOST-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[K]], [[SMIN]]
+; INNERMOST-NEXT:    br i1 [[ITR_CHK1]], label %[[I_HEADER]], label %[[I_EXIT:.*]]
+; INNERMOST:       [[I_EXIT]]:
+; INNERMOST-NEXT:    br label %[[LS_GUARD1]]
+; INNERMOST:       [[LS_GUARD1]]:
+; INNERMOST-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; INNERMOST-NEXT:    br i1 [[ITR_CHK2]], label %[[I_PH_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; INNERMOST:       [[I_PH_LS1]]:
+; INNERMOST-NEXT:    br label %[[I_HEADER_LS1:.*]]
+; INNERMOST:       [[I_HEADER_LS1]]:
+; INNERMOST-NEXT:    [[K_LS1:%.*]] = phi i64 [ [[UMAX]], %[[I_PH_LS1]] ], [ [[K_NEXT_LS1:%.*]], %[[I_HEADER_LS1]] ]
+; INNERMOST-NEXT:    [[IDX_LS1:%.*]] = add i64 [[J]], [[K_LS1]]
+; INNERMOST-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX_LS1]]
+; INNERMOST-NEXT:    store i64 [[IDX_LS1]], ptr [[P_LS1]], align 4
+; INNERMOST-NEXT:    [[K_NEXT_LS1]] = add nsw i64 [[K_LS1]], 1
+; INNERMOST-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[K_LS1]], [[TMP0]]
+; INNERMOST-NEXT:    br i1 [[ITR_CHK3]], label %[[I_HEADER_LS1]], label %[[LS_EXIT1:.*]]
+; INNERMOST:       [[LS_EXIT1]]:
+; INNERMOST-NEXT:    br label %[[LS_FINAL_EXIT]]
+; INNERMOST:       [[LS_FINAL_EXIT]]:
+; INNERMOST-NEXT:    br label %[[M_LATCH]]
+; INNERMOST:       [[M_LATCH]]:
+; INNERMOST-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; INNERMOST-NEXT:    [[JC:%.*]] = icmp slt i64 [[J_NEXT]], [[N]]
+; INNERMOST-NEXT:    br i1 [[JC]], label %[[M_HEADER]], label %[[M_EXIT:.*]]
+; INNERMOST:       [[M_EXIT]]:
+; INNERMOST-NEXT:    br label %[[O_LATCH]]
+; INNERMOST:       [[O_LATCH]]:
+; INNERMOST-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; INNERMOST-NEXT:    [[IC:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; INNERMOST-NEXT:    br i1 [[IC]], label %[[O_HEADER]], label %[[EXIT:.*]]
+; INNERMOST:       [[EXIT]]:
+; INNERMOST-NEXT:    ret void
+;
+; DEPTH1-LABEL: define void @nest3(
+; DEPTH1-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; DEPTH1-NEXT:  [[LS_GUARD0:.*:]]
+; DEPTH1-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; DEPTH1-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; DEPTH1-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
+; DEPTH1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; DEPTH1-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; DEPTH1-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; DEPTH1-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; DEPTH1:       [[ENTRY]]:
+; DEPTH1-NEXT:    br label %[[O_HEADER:.*]]
+; DEPTH1:       [[O_HEADER]]:
+; DEPTH1-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[O_LATCH:.*]] ]
+; DEPTH1-NEXT:    br label %[[M_PH:.*]]
+; DEPTH1:       [[M_PH]]:
+; DEPTH1-NEXT:    br label %[[M_HEADER:.*]]
+; DEPTH1:       [[M_HEADER]]:
+; DEPTH1-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[M_PH]] ], [ [[J_NEXT:%.*]], %[[M_LATCH:.*]] ]
+; DEPTH1-NEXT:    br label %[[I_PH:.*]]
+; DEPTH1:       [[I_PH]]:
+; DEPTH1-NEXT:    br label %[[I_HEADER:.*]]
+; DEPTH1:       [[I_HEADER]]:
+; DEPTH1-NEXT:    [[K:%.*]] = phi i64 [ 0, %[[I_PH]] ], [ [[K_NEXT:%.*]], %[[I_HEADER]] ]
+; DEPTH1-NEXT:    [[IDX:%.*]] = add i64 [[J]], [[K]]
+; DEPTH1-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX]]
+; DEPTH1-NEXT:    store i64 [[IDX]], ptr [[P]], align 4
+; DEPTH1-NEXT:    [[K_NEXT]] = add nsw i64 [[K]], 1
+; DEPTH1-NEXT:    [[KC:%.*]] = icmp slt i64 [[K_NEXT]], [[N]]
+; DEPTH1-NEXT:    br i1 [[KC]], label %[[I_HEADER]], label %[[I_EXIT:.*]]
+; DEPTH1:       [[I_EXIT]]:
+; DEPTH1-NEXT:    br label %[[M_LATCH]]
+; DEPTH1:       [[M_LATCH]]:
+; DEPTH1-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; DEPTH1-NEXT:    [[JC:%.*]] = icmp slt i64 [[J_NEXT]], [[N]]
+; DEPTH1-NEXT:    br i1 [[JC]], label %[[M_HEADER]], label %[[M_EXIT:.*]]
+; DEPTH1:       [[M_EXIT]]:
+; DEPTH1-NEXT:    br label %[[O_LATCH]]
+; DEPTH1:       [[O_LATCH]]:
+; DEPTH1-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; DEPTH1-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[I]], [[SMIN]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK1]], label %[[O_HEADER]], label %[[EXIT:.*]]
+; DEPTH1:       [[EXIT]]:
+; DEPTH1-NEXT:    br label %[[LS_GUARD1]]
+; DEPTH1:       [[LS_GUARD1]]:
+; DEPTH1-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK2]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; DEPTH1:       [[ENTRY_LS1]]:
+; DEPTH1-NEXT:    br label %[[O_HEADER_LS1:.*]]
+; DEPTH1:       [[O_HEADER_LS1]]:
+; DEPTH1-NEXT:    [[I_LS1:%.*]] = phi i64 [ [[UMAX]], %[[ENTRY_LS1]] ], [ [[I_NEXT_LS1:%.*]], %[[O_LATCH_LS1:.*]] ]
+; DEPTH1-NEXT:    br label %[[M_PH_LS1:.*]]
+; DEPTH1:       [[M_PH_LS1]]:
+; DEPTH1-NEXT:    br label %[[M_HEADER_LS1:.*]]
+; DEPTH1:       [[M_HEADER_LS1]]:
+; DEPTH1-NEXT:    [[J_LS1:%.*]] = phi i64 [ 0, %[[M_PH_LS1]] ], [ [[J_NEXT_LS1:%.*]], %[[M_LATCH_LS1:.*]] ]
+; DEPTH1-NEXT:    br label %[[I_PH_LS1:.*]]
+; DEPTH1:       [[I_PH_LS1]]:
+; DEPTH1-NEXT:    br label %[[I_HEADER_LS1:.*]]
+; DEPTH1:       [[I_HEADER_LS1]]:
+; DEPTH1-NEXT:    [[K_LS1:%.*]] = phi i64 [ 0, %[[I_PH_LS1]] ], [ [[K_NEXT_LS1:%.*]], %[[I_HEADER_LS1]] ]
+; DEPTH1-NEXT:    [[IDX_LS1:%.*]] = add i64 [[J_LS1]], [[K_LS1]]
+; DEPTH1-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX_LS1]]
+; DEPTH1-NEXT:    store i64 [[IDX_LS1]], ptr [[P_LS1]], align 4
+; DEPTH1-NEXT:    [[K_NEXT_LS1]] = add nsw i64 [[K_LS1]], 1
+; DEPTH1-NEXT:    [[KC_LS1:%.*]] = icmp slt i64 [[K_NEXT_LS1]], [[N]]
+; DEPTH1-NEXT:    br i1 [[KC_LS1]], label %[[I_HEADER_LS1]], label %[[I_EXIT_LS1:.*]]
+; DEPTH1:       [[I_EXIT_LS1]]:
+; DEPTH1-NEXT:    br label %[[M_LATCH_LS1]]
+; DEPTH1:       [[M_LATCH_LS1]]:
+; DEPTH1-NEXT:    [[J_NEXT_LS1]] = add nsw i64 [[J_LS1]], 1
+; DEPTH1-NEXT:    [[JC_LS1:%.*]] = icmp slt i64 [[J_NEXT_LS1]], [[N]]
+; DEPTH1-NEXT:    br i1 [[JC_LS1]], label %[[M_HEADER_LS1]], label %[[M_EXIT_LS1:.*]]
+; DEPTH1:       [[M_EXIT_LS1]]:
+; DEPTH1-NEXT:    br label %[[O_LATCH_LS1]]
+; DEPTH1:       [[O_LATCH_LS1]]:
+; DEPTH1-NEXT:    [[I_NEXT_LS1]] = add nsw i64 [[I_LS1]], 1
+; DEPTH1-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[I_LS1]], [[TMP0]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK3]], label %[[O_HEADER_LS1]], label %[[LS_EXIT1:.*]]
+; DEPTH1:       [[LS_EXIT1]]:
+; DEPTH1-NEXT:    br label %[[LS_FINAL_EXIT]]
+; DEPTH1:       [[LS_FINAL_EXIT]]:
+; DEPTH1-NEXT:    ret void
+;
+; DEPTH2-LABEL: define void @nest3(
+; DEPTH2-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; DEPTH2-NEXT:  [[ENTRY:.*]]:
+; DEPTH2-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; DEPTH2-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; DEPTH2-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
+; DEPTH2-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; DEPTH2-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; DEPTH2-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; DEPTH2-NEXT:    br label %[[O_HEADER:.*]]
+; DEPTH2:       [[O_HEADER]]:
+; DEPTH2-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[O_LATCH:.*]] ]
+; DEPTH2-NEXT:    br label %[[LS_GUARD0:.*]]
+; DEPTH2:       [[LS_GUARD0]]:
+; DEPTH2-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; DEPTH2-NEXT:    br i1 [[ITR_CHK]], label %[[M_PH:.*]], label %[[LS_GUARD1:.*]]
+; DEPTH2:       [[M_PH]]:
+; DEPTH2-NEXT:    br label %[[M_HEADER:.*]]
+; DEPTH2:       [[M_HEADER]]:
+; DEPTH2-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[M_PH]] ], [ [[J_NEXT:%.*]], %[[M_LATCH:.*]] ]
+; DEPTH2-NEXT:    br label %[[I_PH:.*]]
+; DEPTH2:       [[I_PH]]:
+; DEPTH2-NEXT:    br label %[[I_HEADER:.*]]
+; DEPTH2:       [[I_HEADER]]:
+; DEPTH2-NEXT:    [[K:%.*]] = phi i64 [ 0, %[[I_PH]] ], [ [[K_NEXT:%.*]], %[[I_HEADER]] ]
+; DEPTH2-NEXT:    [[IDX:%.*]] = add i64 [[J]], [[K]]
+; DEPTH2-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX]]
+; DEPTH2-NEXT:    store i64 [[IDX]], ptr [[P]], align 4
+; DEPTH2-NEXT:    [[K_NEXT]] = add nsw i64 [[K]], 1
+; DEPTH2-NEXT:    [[KC:%.*]] = icmp slt i64 [[K_NEXT]], [[N]]
+; DEPTH2-NEXT:    br i1 [[KC]], label %[[I_HEADER]], label %[[I_EXIT:.*]]
+; DEPTH2:       [[I_EXIT]]:
+; DEPTH2-NEXT:    br label %[[M_LATCH]]
+; DEPTH2:       [[M_LATCH]]:
+; DEPTH2-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; DEPTH2-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[J]], [[SMIN]]
+; DEPTH2-NEXT:    br i1 [[ITR_CHK1]], label %[[M_HEADER]], label %[[M_EXIT:.*]]
+; DEPTH2:       [[M_EXIT]]:
+; DEPTH2-NEXT:    br label %[[LS_GUARD1]]
+; DEPTH2:       [[LS_GUARD1]]:
+; DEPTH2-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; DEPTH2-NEXT:    br i1 [[ITR_CHK2]], label %[[M_PH_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; DEPTH2:       [[M_PH_LS1]]:
+; DEPTH2-NEXT:    br label %[[M_HEADER_LS1:.*]]
+; DEPTH2:       [[M_HEADER_LS1]]:
+; DEPTH2-NEXT:    [[J_LS1:%.*]] = phi i64 [ [[UMAX]], %[[M_PH_LS1]] ], [ [[J_NEXT_LS1:%.*]], %[[M_LATCH_LS1:.*]] ]
+; DEPTH2-NEXT:    br label %[[I_PH_LS1:.*]]
+; DEPTH2:       [[I_PH_LS1]]:
+; DEPTH2-NEXT:    br label %[[I_HEADER_LS1:.*]]
+; DEPTH2:       [[I_HEADER_LS1]]:
+; DEPTH2-NEXT:    [[K_LS1:%.*]] = phi i64 [ 0, %[[I_PH_LS1]] ], [ [[K_NEXT_LS1:%.*]], %[[I_HEADER_LS1]] ]
+; DEPTH2-NEXT:    [[IDX_LS1:%.*]] = add i64 [[J_LS1]], [[K_LS1]]
+; DEPTH2-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IDX_LS1]]
+; DEPTH2-NEXT:    store i64 [[IDX_LS1]], ptr [[P_LS1]], align 4
+; DEPTH2-NEXT:    [[K_NEXT_LS1]] = add nsw i64 [[K_LS1]], 1
+; DEPTH2-NEXT:    [[KC_LS1:%.*]] = icmp slt i64 [[K_NEXT_LS1]], [[N]]
+; DEPTH2-NEXT:    br i1 [[KC_LS1]], label %[[I_HEADER_LS1]], label %[[I_EXIT_LS1:.*]]
+; DEPTH2:       [[I_EXIT_LS1]]:
+; DEPTH2-NEXT:    br label %[[M_LATCH_LS1]]
+; DEPTH2:       [[M_LATCH_LS1]]:
+; DEPTH2-NEXT:    [[J_NEXT_LS1]] = add nsw i64 [[J_LS1]], 1
+; DEPTH2-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[J_LS1]], [[TMP0]]
+; DEPTH2-NEXT:    br i1 [[ITR_CHK3]], label %[[M_HEADER_LS1]], label %[[LS_EXIT1:.*]]
+; DEPTH2:       [[LS_EXIT1]]:
+; DEPTH2-NEXT:    br label %[[LS_FINAL_EXIT]]
+; DEPTH2:       [[LS_FINAL_EXIT]]:
+; DEPTH2-NEXT:    br label %[[O_LATCH]]
+; DEPTH2:       [[O_LATCH]]:
+; DEPTH2-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; DEPTH2-NEXT:    [[IC:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; DEPTH2-NEXT:    br i1 [[IC]], label %[[O_HEADER]], label %[[EXIT:.*]]
+; DEPTH2:       [[EXIT]]:
+; DEPTH2-NEXT:    ret void
+;
+entry:
+  br label %o.header
+
+o.header:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %o.latch ]
+  br label %m.ph
+
+m.ph:
+  br label %m.header
+
+m.header:
+  %j = phi i64 [ 0, %m.ph ], [ %j.next, %m.latch ]
+  br label %i.ph
+
+i.ph:
+  br label %i.header
+
+i.header:
+  %k = phi i64 [ 0, %i.ph ], [ %k.next, %i.header ]
+  %idx = add i64 %j, %k
+  %p = getelementptr inbounds i64, ptr %a, i64 %idx
+  store i64 %idx, ptr %p
+  %k.next = add nsw i64 %k, 1
+  %kc = icmp slt i64 %k.next, %n
+  br i1 %kc, label %i.header, label %i.exit
+
+i.exit:
+  br label %m.latch
+
+m.latch:
+  %j.next = add nsw i64 %j, 1
+  %jc = icmp slt i64 %j.next, %n
+  br i1 %jc, label %m.header, label %m.exit
+
+m.exit:
+  br label %o.latch
+
+o.latch:
+  %i.next = add nsw i64 %i, 1
+  %ic = icmp slt i64 %i.next, %n
+  br i1 %ic, label %o.header, label %exit
+
+exit:
+  ret void
+}
+
+; The inner loop carries %acc, so it is rejected, but the outer loop carries
+; nothing and splits at depth 1. Selecting a different depth reaches a loop the
+; default never offers to the utility.
+
+define void @inner_carried_outer_clean(ptr %a, i64 %n, i64 %m) {
+; INNERMOST-LABEL: define void @inner_carried_outer_clean(
+; INNERMOST-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; INNERMOST-NEXT:  [[ENTRY:.*]]:
+; INNERMOST-NEXT:    br label %[[OUTER_HEADER:.*]]
+; INNERMOST:       [[OUTER_HEADER]]:
+; INNERMOST-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; INNERMOST-NEXT:    br label %[[INNER_PH:.*]]
+; INNERMOST:       [[INNER_PH]]:
+; INNERMOST-NEXT:    br label %[[INNER_HEADER:.*]]
+; INNERMOST:       [[INNER_HEADER]]:
+; INNERMOST-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[INNER_PH]] ], [ [[J_NEXT:%.*]], %[[INNER_HEADER]] ]
+; INNERMOST-NEXT:    [[ACC:%.*]] = phi i64 [ [[I]], %[[INNER_PH]] ], [ [[ACC_NEXT:%.*]], %[[INNER_HEADER]] ]
+; INNERMOST-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[J]]
+; INNERMOST-NEXT:    store i64 [[ACC]], ptr [[P]], align 4
+; INNERMOST-NEXT:    [[ACC_NEXT]] = add i64 [[ACC]], 3
+; INNERMOST-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; INNERMOST-NEXT:    [[IC:%.*]] = icmp slt i64 [[J_NEXT]], [[M]]
+; INNERMOST-NEXT:    br i1 [[IC]], label %[[INNER_HEADER]], label %[[INNER_EXIT:.*]]
+; INNERMOST:       [[INNER_EXIT]]:
+; INNERMOST-NEXT:    br label %[[OUTER_LATCH]]
+; INNERMOST:       [[OUTER_LATCH]]:
+; INNERMOST-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; INNERMOST-NEXT:    [[OC:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; INNERMOST-NEXT:    br i1 [[OC]], label %[[OUTER_HEADER]], label %[[EXIT:.*]]
+; INNERMOST:       [[EXIT]]:
+; INNERMOST-NEXT:    ret void
+;
+; DEPTH1-LABEL: define void @inner_carried_outer_clean(
+; DEPTH1-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; DEPTH1-NEXT:  [[LS_GUARD0:.*:]]
+; DEPTH1-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; DEPTH1-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; DEPTH1-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
+; DEPTH1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; DEPTH1-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; DEPTH1-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; DEPTH1-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; DEPTH1:       [[ENTRY]]:
+; DEPTH1-NEXT:    br label %[[OUTER_HEADER:.*]]
+; DEPTH1:       [[OUTER_HEADER]]:
+; DEPTH1-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; DEPTH1-NEXT:    br label %[[INNER_PH:.*]]
+; DEPTH1:       [[INNER_PH]]:
+; DEPTH1-NEXT:    br label %[[INNER_HEADER:.*]]
+; DEPTH1:       [[INNER_HEADER]]:
+; DEPTH1-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[INNER_PH]] ], [ [[J_NEXT:%.*]], %[[INNER_HEADER]] ]
+; DEPTH1-NEXT:    [[ACC:%.*]] = phi i64 [ [[I]], %[[INNER_PH]] ], [ [[ACC_NEXT:%.*]], %[[INNER_HEADER]] ]
+; DEPTH1-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[J]]
+; DEPTH1-NEXT:    store i64 [[ACC]], ptr [[P]], align 4
+; DEPTH1-NEXT:    [[ACC_NEXT]] = add i64 [[ACC]], 3
+; DEPTH1-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; DEPTH1-NEXT:    [[IC:%.*]] = icmp slt i64 [[J_NEXT]], [[M]]
+; DEPTH1-NEXT:    br i1 [[IC]], label %[[INNER_HEADER]], label %[[INNER_EXIT:.*]]
+; DEPTH1:       [[INNER_EXIT]]:
+; DEPTH1-NEXT:    br label %[[OUTER_LATCH]]
+; DEPTH1:       [[OUTER_LATCH]]:
+; DEPTH1-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; DEPTH1-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[I]], [[SMIN]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK1]], label %[[OUTER_HEADER]], label %[[EXIT:.*]]
+; DEPTH1:       [[EXIT]]:
+; DEPTH1-NEXT:    br label %[[LS_GUARD1]]
+; DEPTH1:       [[LS_GUARD1]]:
+; DEPTH1-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK2]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; DEPTH1:       [[ENTRY_LS1]]:
+; DEPTH1-NEXT:    br label %[[OUTER_HEADER_LS1:.*]]
+; DEPTH1:       [[OUTER_HEADER_LS1]]:
+; DEPTH1-NEXT:    [[I_LS1:%.*]] = phi i64 [ [[UMAX]], %[[ENTRY_LS1]] ], [ [[I_NEXT_LS1:%.*]], %[[OUTER_LATCH_LS1:.*]] ]
+; DEPTH1-NEXT:    br label %[[INNER_PH_LS1:.*]]
+; DEPTH1:       [[INNER_PH_LS1]]:
+; DEPTH1-NEXT:    br label %[[INNER_HEADER_LS1:.*]]
+; DEPTH1:       [[INNER_HEADER_LS1]]:
+; DEPTH1-NEXT:    [[J_LS1:%.*]] = phi i64 [ 0, %[[INNER_PH_LS1]] ], [ [[J_NEXT_LS1:%.*]], %[[INNER_HEADER_LS1]] ]
+; DEPTH1-NEXT:    [[ACC_LS1:%.*]] = phi i64 [ [[I_LS1]], %[[INNER_PH_LS1]] ], [ [[ACC_NEXT_LS1:%.*]], %[[INNER_HEADER_LS1]] ]
+; DEPTH1-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[J_LS1]]
+; DEPTH1-NEXT:    store i64 [[ACC_LS1]], ptr [[P_LS1]], align 4
+; DEPTH1-NEXT:    [[ACC_NEXT_LS1]] = add i64 [[ACC_LS1]], 3
+; DEPTH1-NEXT:    [[J_NEXT_LS1]] = add nsw i64 [[J_LS1]], 1
+; DEPTH1-NEXT:    [[IC_LS1:%.*]] = icmp slt i64 [[J_NEXT_LS1]], [[M]]
+; DEPTH1-NEXT:    br i1 [[IC_LS1]], label %[[INNER_HEADER_LS1]], label %[[INNER_EXIT_LS1:.*]]
+; DEPTH1:       [[INNER_EXIT_LS1]]:
+; DEPTH1-NEXT:    br label %[[OUTER_LATCH_LS1]]
+; DEPTH1:       [[OUTER_LATCH_LS1]]:
+; DEPTH1-NEXT:    [[I_NEXT_LS1]] = add nsw i64 [[I_LS1]], 1
+; DEPTH1-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[I_LS1]], [[TMP0]]
+; DEPTH1-NEXT:    br i1 [[ITR_CHK3]], label %[[OUTER_HEADER_LS1]], label %[[LS_EXIT1:.*]]
+; DEPTH1:       [[LS_EXIT1]]:
+; DEPTH1-NEXT:    br label %[[LS_FINAL_EXIT]]
+; DEPTH1:       [[LS_FINAL_EXIT]]:
+; DEPTH1-NEXT:    ret void
+;
+; DEPTH2-LABEL: define void @inner_carried_outer_clean(
+; DEPTH2-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i64 [[M:%.*]]) {
+; DEPTH2-NEXT:  [[ENTRY:.*]]:
+; DEPTH2-NEXT:    br label %[[OUTER_HEADER:.*]]
+; DEPTH2:       [[OUTER_HEADER]]:
+; DEPTH2-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; DEPTH2-NEXT:    br label %[[INNER_PH:.*]]
+; DEPTH2:       [[INNER_PH]]:
+; DEPTH2-NEXT:    br label %[[INNER_HEADER:.*]]
+; DEPTH2:       [[INNER_HEADER]]:
+; DEPTH2-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[INNER_PH]] ], [ [[J_NEXT:%.*]], %[[INNER_HEADER]] ]
+; DEPTH2-NEXT:    [[ACC:%.*]] = phi i64 [ [[I]], %[[INNER_PH]] ], [ [[ACC_NEXT:%.*]], %[[INNER_HEADER]] ]
+; DEPTH2-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[J]]
+; DEPTH2-NEXT:    store i64 [[ACC]], ptr [[P]], align 4
+; DEPTH2-NEXT:    [[ACC_NEXT]] = add i64 [[ACC]], 3
+; DEPTH2-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; DEPTH2-NEXT:    [[IC:%.*]] = icmp slt i64 [[J_NEXT]], [[M]]
+; DEPTH2-NEXT:    br i1 [[IC]], label %[[INNER_HEADER]], label %[[INNER_EXIT:.*]]
+; DEPTH2:       [[INNER_EXIT]]:
+; DEPTH2-NEXT:    br label %[[OUTER_LATCH]]
+; DEPTH2:       [[OUTER_LATCH]]:
+; DEPTH2-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; DEPTH2-NEXT:    [[OC:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; DEPTH2-NEXT:    br i1 [[OC]], label %[[OUTER_HEADER]], label %[[EXIT:.*]]
+; DEPTH2:       [[EXIT]]:
+; DEPTH2-NEXT:    ret void
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %outer.latch ]
+  br label %inner.ph
+
+inner.ph:
+  br label %inner.header
+
+inner.header:
+  %j = phi i64 [ 0, %inner.ph ], [ %j.next, %inner.header ]
+  %acc = phi i64 [ %i, %inner.ph ], [ %acc.next, %inner.header ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %j
+  store i64 %acc, ptr %p
+  %acc.next = add i64 %acc, 3
+  %j.next = add nsw i64 %j, 1
+  %ic = icmp slt i64 %j.next, %m
+  br i1 %ic, label %inner.header, label %inner.exit
+
+inner.exit:
+  br label %outer.latch
+
+outer.latch:
+  %i.next = add nsw i64 %i, 1
+  %oc = icmp slt i64 %i.next, %n
+  br i1 %oc, label %outer.header, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopSplit/multiple-partitions.ll b/llvm/test/Transforms/LoopSplit/multiple-partitions.ll
new file mode 100644
index 0000000000000..21d068181d040
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/multiple-partitions.ll
@@ -0,0 +1,155 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=10,20 -S < %s | FileCheck %s --check-prefix=THREE
+; RUN: opt -passes=loop-split -loop-split-points=10,20,30 -S < %s | FileCheck %s --check-prefix=FOUR
+
+; More than one split point tiles the space into a longer chain. Each guard
+; either enters its partition or falls through to the next guard, and the last
+; one falls through to ls.final.exit.
+
+define void @tile(ptr %a, i64 %n) {
+; THREE-LABEL: define void @tile(
+; THREE-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; THREE-NEXT:  [[LS_GUARD0:.*:]]
+; THREE-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; THREE-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; THREE-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 10)
+; THREE-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; THREE-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; THREE-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; THREE-NEXT:    [[UMIN1:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 20)
+; THREE-NEXT:    [[UMAX2:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN1]], i64 1)
+; THREE-NEXT:    [[TMP2:%.*]] = add nsw i64 [[UMAX2]], -1
+; THREE-NEXT:    [[SMIN1:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP2]])
+; THREE-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; THREE-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; THREE:       [[ENTRY]]:
+; THREE-NEXT:    br label %[[LOOP:.*]]
+; THREE:       [[LOOP]]:
+; THREE-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; THREE-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; THREE-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; THREE-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; THREE-NEXT:    [[ITR_CHK2:%.*]] = icmp slt i64 [[IV]], [[SMIN]]
+; THREE-NEXT:    br i1 [[ITR_CHK2]], label %[[LOOP]], label %[[EXIT:.*]]
+; THREE:       [[EXIT]]:
+; THREE-NEXT:    br label %[[LS_GUARD1]]
+; THREE:       [[LS_GUARD1]]:
+; THREE-NEXT:    [[ITR_CHK3:%.*]] = icmp sle i64 [[UMAX]], [[SMIN1]]
+; THREE-NEXT:    br i1 [[ITR_CHK3]], label %[[ENTRY_LS1:.*]], label %[[LS_GUARD2:.*]]
+; THREE:       [[ENTRY_LS1]]:
+; THREE-NEXT:    br label %[[LOOP_LS1:.*]]
+; THREE:       [[LOOP_LS1]]:
+; THREE-NEXT:    [[IV_LS1:%.*]] = phi i64 [ [[UMAX]], %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; THREE-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; THREE-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; THREE-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], 1
+; THREE-NEXT:    [[ITR_CHK4:%.*]] = icmp slt i64 [[IV_LS1]], [[SMIN1]]
+; THREE-NEXT:    br i1 [[ITR_CHK4]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; THREE:       [[LS_EXIT1]]:
+; THREE-NEXT:    br label %[[LS_GUARD2]]
+; THREE:       [[LS_GUARD2]]:
+; THREE-NEXT:    [[ITR_CHK5:%.*]] = icmp sle i64 [[UMAX2]], [[TMP0]]
+; THREE-NEXT:    br i1 [[ITR_CHK5]], label %[[ENTRY_LS2:.*]], label %[[LS_FINAL_EXIT:.*]]
+; THREE:       [[ENTRY_LS2]]:
+; THREE-NEXT:    br label %[[LOOP_LS2:.*]]
+; THREE:       [[LOOP_LS2]]:
+; THREE-NEXT:    [[IV_LS2:%.*]] = phi i64 [ [[UMAX2]], %[[ENTRY_LS2]] ], [ [[IV_NEXT_LS2:%.*]], %[[LOOP_LS2]] ]
+; THREE-NEXT:    [[P_LS2:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS2]]
+; THREE-NEXT:    store i64 [[IV_LS2]], ptr [[P_LS2]], align 4
+; THREE-NEXT:    [[IV_NEXT_LS2]] = add nsw i64 [[IV_LS2]], 1
+; THREE-NEXT:    [[ITR_CHK6:%.*]] = icmp slt i64 [[IV_LS2]], [[TMP0]]
+; THREE-NEXT:    br i1 [[ITR_CHK6]], label %[[LOOP_LS2]], label %[[LS_EXIT2:.*]]
+; THREE:       [[LS_EXIT2]]:
+; THREE-NEXT:    br label %[[LS_FINAL_EXIT]]
+; THREE:       [[LS_FINAL_EXIT]]:
+; THREE-NEXT:    ret void
+;
+; FOUR-LABEL: define void @tile(
+; FOUR-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; FOUR-NEXT:  [[LS_GUARD0:.*:]]
+; FOUR-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; FOUR-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; FOUR-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 10)
+; FOUR-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; FOUR-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; FOUR-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; FOUR-NEXT:    [[UMIN1:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 20)
+; FOUR-NEXT:    [[UMAX2:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN1]], i64 1)
+; FOUR-NEXT:    [[TMP2:%.*]] = add nsw i64 [[UMAX2]], -1
+; FOUR-NEXT:    [[SMIN1:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP2]])
+; FOUR-NEXT:    [[UMIN4:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 30)
+; FOUR-NEXT:    [[UMAX5:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN4]], i64 1)
+; FOUR-NEXT:    [[TMP3:%.*]] = add nsw i64 [[UMAX5]], -1
+; FOUR-NEXT:    [[SMIN2:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP3]])
+; FOUR-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; FOUR-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; FOUR:       [[ENTRY]]:
+; FOUR-NEXT:    br label %[[LOOP:.*]]
+; FOUR:       [[LOOP]]:
+; FOUR-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; FOUR-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; FOUR-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; FOUR-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; FOUR-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[IV]], [[SMIN]]
+; FOUR-NEXT:    br i1 [[ITR_CHK3]], label %[[LOOP]], label %[[EXIT:.*]]
+; FOUR:       [[EXIT]]:
+; FOUR-NEXT:    br label %[[LS_GUARD1]]
+; FOUR:       [[LS_GUARD1]]:
+; FOUR-NEXT:    [[ITR_CHK4:%.*]] = icmp sle i64 [[UMAX]], [[SMIN1]]
+; FOUR-NEXT:    br i1 [[ITR_CHK4]], label %[[ENTRY_LS1:.*]], label %[[LS_GUARD2:.*]]
+; FOUR:       [[ENTRY_LS1]]:
+; FOUR-NEXT:    br label %[[LOOP_LS1:.*]]
+; FOUR:       [[LOOP_LS1]]:
+; FOUR-NEXT:    [[IV_LS1:%.*]] = phi i64 [ [[UMAX]], %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; FOUR-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; FOUR-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; FOUR-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], 1
+; FOUR-NEXT:    [[ITR_CHK5:%.*]] = icmp slt i64 [[IV_LS1]], [[SMIN1]]
+; FOUR-NEXT:    br i1 [[ITR_CHK5]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; FOUR:       [[LS_EXIT1]]:
+; FOUR-NEXT:    br label %[[LS_GUARD2]]
+; FOUR:       [[LS_GUARD2]]:
+; FOUR-NEXT:    [[ITR_CHK6:%.*]] = icmp sle i64 [[UMAX2]], [[SMIN2]]
+; FOUR-NEXT:    br i1 [[ITR_CHK6]], label %[[ENTRY_LS2:.*]], label %[[LS_GUARD3:.*]]
+; FOUR:       [[ENTRY_LS2]]:
+; FOUR-NEXT:    br label %[[LOOP_LS2:.*]]
+; FOUR:       [[LOOP_LS2]]:
+; FOUR-NEXT:    [[IV_LS2:%.*]] = phi i64 [ [[UMAX2]], %[[ENTRY_LS2]] ], [ [[IV_NEXT_LS2:%.*]], %[[LOOP_LS2]] ]
+; FOUR-NEXT:    [[P_LS2:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS2]]
+; FOUR-NEXT:    store i64 [[IV_LS2]], ptr [[P_LS2]], align 4
+; FOUR-NEXT:    [[IV_NEXT_LS2]] = add nsw i64 [[IV_LS2]], 1
+; FOUR-NEXT:    [[ITR_CHK7:%.*]] = icmp slt i64 [[IV_LS2]], [[SMIN2]]
+; FOUR-NEXT:    br i1 [[ITR_CHK7]], label %[[LOOP_LS2]], label %[[LS_EXIT2:.*]]
+; FOUR:       [[LS_EXIT2]]:
+; FOUR-NEXT:    br label %[[LS_GUARD3]]
+; FOUR:       [[LS_GUARD3]]:
+; FOUR-NEXT:    [[ITR_CHK8:%.*]] = icmp sle i64 [[UMAX5]], [[TMP0]]
+; FOUR-NEXT:    br i1 [[ITR_CHK8]], label %[[ENTRY_LS3:.*]], label %[[LS_FINAL_EXIT:.*]]
+; FOUR:       [[ENTRY_LS3]]:
+; FOUR-NEXT:    br label %[[LOOP_LS3:.*]]
+; FOUR:       [[LOOP_LS3]]:
+; FOUR-NEXT:    [[IV_LS3:%.*]] = phi i64 [ [[UMAX5]], %[[ENTRY_LS3]] ], [ [[IV_NEXT_LS3:%.*]], %[[LOOP_LS3]] ]
+; FOUR-NEXT:    [[P_LS3:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS3]]
+; FOUR-NEXT:    store i64 [[IV_LS3]], ptr [[P_LS3]], align 4
+; FOUR-NEXT:    [[IV_NEXT_LS3]] = add nsw i64 [[IV_LS3]], 1
+; FOUR-NEXT:    [[ITR_CHK9:%.*]] = icmp slt i64 [[IV_LS3]], [[TMP0]]
+; FOUR-NEXT:    br i1 [[ITR_CHK9]], label %[[LOOP_LS3]], label %[[LS_EXIT3:.*]]
+; FOUR:       [[LS_EXIT3]]:
+; FOUR-NEXT:    br label %[[LS_FINAL_EXIT]]
+; FOUR:       [[LS_FINAL_EXIT]]:
+; FOUR-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nsw i64 %iv, 1
+  %c = icmp slt i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopSplit/oversized-split-offset.ll b/llvm/test/Transforms/LoopSplit/oversized-split-offset.ll
new file mode 100644
index 0000000000000..4f8931459138b
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/oversized-split-offset.ll
@@ -0,0 +1,87 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=1000,2000 -S < %s | FileCheck %s --check-prefix=WIDE
+; RUN: opt -passes=loop-split -loop-split-points=100,200 -S < %s | FileCheck %s --check-prefix=NARROW
+
+; Split offsets are sorted so the partition boundaries come out in iteration
+; order, but they are then interpreted in the induction type. For this i8
+; induction 1000 and 2000 truncate to 232 and 208, reversing the order that was
+; just established, so the partitions would overlap instead of tiling and
+; iterations would run twice. Offsets that do not fit the induction type are
+; dropped, leaving nothing to split on.
+
+; The same loop with offsets that do fit is split normally, so the offsets are
+; the only thing standing in the way.
+
+define void @i8_induction(ptr %a) {
+; WIDE-LABEL: define void @i8_induction(
+; WIDE-SAME: ptr [[A:%.*]]) {
+; WIDE-NEXT:  [[ENTRY:.*]]:
+; WIDE-NEXT:    br label %[[LOOP:.*]]
+; WIDE:       [[LOOP]]:
+; WIDE-NEXT:    [[IV:%.*]] = phi i8 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; WIDE-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; WIDE-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; WIDE-NEXT:    [[IV_NEXT]] = add i8 [[IV]], 1
+; WIDE-NEXT:    [[C:%.*]] = icmp ule i8 [[IV]], -27
+; WIDE-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; WIDE:       [[EXIT]]:
+; WIDE-NEXT:    ret void
+;
+; NARROW-LABEL: define void @i8_induction(
+; NARROW-SAME: ptr [[A:%.*]]) {
+; NARROW-NEXT:  [[LS_GUARD0:.*:]]
+; NARROW-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; NARROW:       [[ENTRY]]:
+; NARROW-NEXT:    br label %[[LOOP:.*]]
+; NARROW:       [[LOOP]]:
+; NARROW-NEXT:    [[IV:%.*]] = phi i8 [ 1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; NARROW-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; NARROW-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; NARROW-NEXT:    [[IV_NEXT]] = add i8 [[IV]], 1
+; NARROW-NEXT:    [[ITR_CHK:%.*]] = icmp ult i8 [[IV]], 100
+; NARROW-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
+; NARROW:       [[EXIT]]:
+; NARROW-NEXT:    br label %[[LS_GUARD1]]
+; NARROW:       [[LS_GUARD1]]:
+; NARROW-NEXT:    br i1 true, label %[[ENTRY_LS1:.*]], label %[[LS_GUARD2:.*]]
+; NARROW:       [[ENTRY_LS1]]:
+; NARROW-NEXT:    br label %[[LOOP_LS1:.*]]
+; NARROW:       [[LOOP_LS1]]:
+; NARROW-NEXT:    [[IV_LS1:%.*]] = phi i8 [ 101, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; NARROW-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV_LS1]]
+; NARROW-NEXT:    store i8 [[IV_LS1]], ptr [[P_LS1]], align 1
+; NARROW-NEXT:    [[IV_NEXT_LS1]] = add i8 [[IV_LS1]], 1
+; NARROW-NEXT:    [[ITR_CHK1:%.*]] = icmp ult i8 [[IV_LS1]], -56
+; NARROW-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; NARROW:       [[LS_EXIT1]]:
+; NARROW-NEXT:    br label %[[LS_GUARD2]]
+; NARROW:       [[LS_GUARD2]]:
+; NARROW-NEXT:    br i1 true, label %[[ENTRY_LS2:.*]], label %[[LS_FINAL_EXIT:.*]]
+; NARROW:       [[ENTRY_LS2]]:
+; NARROW-NEXT:    br label %[[LOOP_LS2:.*]]
+; NARROW:       [[LOOP_LS2]]:
+; NARROW-NEXT:    [[IV_LS2:%.*]] = phi i8 [ -55, %[[ENTRY_LS2]] ], [ [[IV_NEXT_LS2:%.*]], %[[LOOP_LS2]] ]
+; NARROW-NEXT:    [[P_LS2:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV_LS2]]
+; NARROW-NEXT:    store i8 [[IV_LS2]], ptr [[P_LS2]], align 1
+; NARROW-NEXT:    [[IV_NEXT_LS2]] = add i8 [[IV_LS2]], 1
+; NARROW-NEXT:    [[ITR_CHK2:%.*]] = icmp ult i8 [[IV_LS2]], -26
+; NARROW-NEXT:    br i1 [[ITR_CHK2]], label %[[LOOP_LS2]], label %[[LS_EXIT2:.*]]
+; NARROW:       [[LS_EXIT2]]:
+; NARROW-NEXT:    br label %[[LS_FINAL_EXIT]]
+; NARROW:       [[LS_FINAL_EXIT]]:
+; NARROW-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ 1, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, 1
+  %c = icmp ule i8 %iv, -27
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopSplit/saturating-latch.ll b/llvm/test/Transforms/LoopSplit/saturating-latch.ll
new file mode 100644
index 0000000000000..35bc5733a3c60
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/saturating-latch.ll
@@ -0,0 +1,222 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=3 -S < %s | FileCheck %s
+
+; Each partition's latch is rebuilt as the strict `i < sel` on the induction PHI
+; rather than the equivalent `i + 1 <= sel` on the step value. Both express the
+; same test -- keep going while the value the next iteration would use is still
+; in the partition -- but only the strict form avoids computing `i + 1`, so it
+; stays a real test when `sel` is the last value of the iteration direction. The
+; inclusive form saturates there and the partition would never exit.
+;
+; Every loop below ends on such a value, so each one relies on that. Note the
+; last partition's latch compares against the extreme itself and still
+; terminates.
+
+; The induction walks every i8, so the end is 255, the unsigned maximum. The
+; original latch compares the step value, which is what reaches the extreme: at
+; 255 the increment wraps to 0 and the `ne` test fails.
+
+define void @ascending_reaches_umax(ptr %a) {
+; CHECK-LABEL: define void @ascending_reaches_umax(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i8 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp ult i8 [[IV]], 2
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    br i1 true, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i8 [ 3, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV_LS1]]
+; CHECK-NEXT:    store i8 [[IV_LS1]], ptr [[P_LS1]], align 1
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add i8 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp ult i8 [[IV_LS1]], -1
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, 1
+  %c = icmp ne i8 %iv.next, 0
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The same for a signed ordering, ending on 127, the signed maximum.
+
+define void @ascending_reaches_smax(ptr %a) {
+; CHECK-LABEL: define void @ascending_reaches_smax(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i8 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp slt i8 [[IV]], 2
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    br i1 true, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i8 [ 3, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV_LS1]]
+; CHECK-NEXT:    store i8 [[IV_LS1]], ptr [[P_LS1]], align 1
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add i8 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i8 [[IV_LS1]], 127
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, 1
+  %c = icmp ne i8 %iv.next, -128
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; Descending, ending on -128, the signed minimum. A descending loop can only
+; reach its extreme through a latch comparing the PHI: once the step value wraps
+; the test stops failing, so a latch comparing it never gets there.
+
+define void @descending_reaches_smin(ptr %a) {
+; CHECK-LABEL: define void @descending_reaches_smin(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ 127, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i8 [[IV]], -1
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sgt i8 [[IV]], 125
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    br i1 true, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i8 [ 124, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV_LS1]]
+; CHECK-NEXT:    store i8 [[IV_LS1]], ptr [[P_LS1]], align 1
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add i8 [[IV_LS1]], -1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp sgt i8 [[IV_LS1]], -128
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ 127, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, -1
+  %c = icmp sgt i8 %iv, -128
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; A symbolic end reaches the extreme just as well: nothing rules out %n being 0,
+; which walks the induction all the way to 255. There is no bound to prove
+; anything about, so this only splits because the rebuilt latch cannot saturate.
+
+define void @symbolic_end_may_reach_umax(ptr %a, i8 %n) {
+; CHECK-LABEL: define void @symbolic_end_may_reach_umax(
+; CHECK-SAME: ptr [[A:%.*]], i8 [[N:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    [[TMP0:%.*]] = add i8 [[N]], -1
+; CHECK-NEXT:    [[UMIN:%.*]] = call i8 @llvm.umin.i8(i8 [[TMP0]], i8 3)
+; CHECK-NEXT:    [[UMAX:%.*]] = call i8 @llvm.umax.i8(i8 [[UMIN]], i8 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i8 [[UMAX]], -1
+; CHECK-NEXT:    [[UMIN1:%.*]] = call i8 @llvm.umin.i8(i8 [[TMP0]], i8 [[TMP1]])
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp ule i8 0, [[UMIN1]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i8 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK2:%.*]] = icmp ult i8 [[IV]], [[UMIN1]]
+; CHECK-NEXT:    br i1 [[ITR_CHK2]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp ule i8 [[UMAX]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i8 [ [[UMAX]], %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV_LS1]]
+; CHECK-NEXT:    store i8 [[IV_LS1]], ptr [[P_LS1]], align 1
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add i8 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK4:%.*]] = icmp ult i8 [[IV_LS1]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK4]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, 1
+  %c = icmp ne i8 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopSplit/sequential-loops.ll b/llvm/test/Transforms/LoopSplit/sequential-loops.ll
new file mode 100644
index 0000000000000..00b0122ed2fe0
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/sequential-loops.ll
@@ -0,0 +1,107 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=50 -S < %s | FileCheck %s
+
+; The driver collects its worklist before transforming anything, so splitting one
+; loop must leave the loops that follow it intact and still splittable. The
+; second loop's preheader is the first loop's exit region, which the first split
+; rewrites.
+
+define void @two_loops(ptr %a, i64 %n) {
+; CHECK-LABEL: define void @two_loops(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 50)
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP1:.*]]
+; CHECK:       [[LOOP1]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[I]]
+; CHECK-NEXT:    store i64 [[I]], ptr [[P]], align 4
+; CHECK-NEXT:    [[I_NEXT]] = add nsw i64 [[I]], 1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[I]], [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP1]], label %[[MID:.*]]
+; CHECK:       [[MID]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK2]], label %[[ENTRY_LS1:.*]], label %[[LS_GUARD05:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP1_LS1:.*]]
+; CHECK:       [[LOOP1_LS1]]:
+; CHECK-NEXT:    [[I_LS1:%.*]] = phi i64 [ [[UMAX]], %[[ENTRY_LS1]] ], [ [[I_NEXT_LS1:%.*]], %[[LOOP1_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[I_LS1]]
+; CHECK-NEXT:    store i64 [[I_LS1]], ptr [[P_LS1]], align 4
+; CHECK-NEXT:    [[I_NEXT_LS1]] = add nsw i64 [[I_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[I_LS1]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[LOOP1_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_GUARD05]]
+; CHECK:       [[LS_GUARD05]]:
+; CHECK-NEXT:    [[SMAX6:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP2:%.*]] = add nsw i64 [[SMAX6]], -1
+; CHECK-NEXT:    [[UMIN7:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP2]], i64 50)
+; CHECK-NEXT:    [[UMAX8:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN7]], i64 1)
+; CHECK-NEXT:    [[TMP3:%.*]] = add nsw i64 [[UMAX8]], -1
+; CHECK-NEXT:    [[SMIN9:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP2]], i64 [[TMP3]])
+; CHECK-NEXT:    [[ITR_CHK12:%.*]] = icmp sle i64 0, [[SMIN9]]
+; CHECK-NEXT:    br i1 [[ITR_CHK12]], label %[[LS_FINAL_EXIT:.*]], label %[[LS_GUARD111:.*]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    br label %[[LOOP2:.*]]
+; CHECK:       [[LOOP2]]:
+; CHECK-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[LS_FINAL_EXIT]] ], [ [[J_NEXT:%.*]], %[[LOOP2]] ]
+; CHECK-NEXT:    [[Q:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[J]]
+; CHECK-NEXT:    store i64 [[J]], ptr [[Q]], align 4
+; CHECK-NEXT:    [[J_NEXT]] = add nsw i64 [[J]], 1
+; CHECK-NEXT:    [[ITR_CHK13:%.*]] = icmp slt i64 [[J]], [[SMIN9]]
+; CHECK-NEXT:    br i1 [[ITR_CHK13]], label %[[LOOP2]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD111]]
+; CHECK:       [[LS_GUARD111]]:
+; CHECK-NEXT:    [[ITR_CHK14:%.*]] = icmp sle i64 [[UMAX8]], [[TMP2]]
+; CHECK-NEXT:    br i1 [[ITR_CHK14]], label %[[LS_FINAL_EXIT_LS1:.*]], label %[[LS_FINAL_EXIT4:.*]]
+; CHECK:       [[LS_FINAL_EXIT_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP2_LS1:.*]]
+; CHECK:       [[LOOP2_LS1]]:
+; CHECK-NEXT:    [[J_LS1:%.*]] = phi i64 [ [[UMAX8]], %[[LS_FINAL_EXIT_LS1]] ], [ [[J_NEXT_LS1:%.*]], %[[LOOP2_LS1]] ]
+; CHECK-NEXT:    [[Q_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[J_LS1]]
+; CHECK-NEXT:    store i64 [[J_LS1]], ptr [[Q_LS1]], align 4
+; CHECK-NEXT:    [[J_NEXT_LS1]] = add nsw i64 [[J_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK15:%.*]] = icmp slt i64 [[J_LS1]], [[TMP2]]
+; CHECK-NEXT:    br i1 [[ITR_CHK15]], label %[[LOOP2_LS1]], label %[[LS_EXIT110:.*]]
+; CHECK:       [[LS_EXIT110]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT4]]
+; CHECK:       [[LS_FINAL_EXIT4]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop1
+
+loop1:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop1 ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %i
+  store i64 %i, ptr %p
+  %i.next = add nsw i64 %i, 1
+  %c1 = icmp slt i64 %i.next, %n
+  br i1 %c1, label %loop1, label %mid
+
+mid:
+  br label %loop2
+
+loop2:
+  %j = phi i64 [ 0, %mid ], [ %j.next, %loop2 ]
+  %q = getelementptr inbounds i64, ptr %a, i64 %j
+  store i64 %j, ptr %q
+  %j.next = add nsw i64 %j, 1
+  %c2 = icmp slt i64 %j.next, %n
+  br i1 %c2, label %loop2, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopSplit/unsupported-forms.ll b/llvm/test/Transforms/LoopSplit/unsupported-forms.ll
new file mode 100644
index 0000000000000..57351abae931c
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/unsupported-forms.ll
@@ -0,0 +1,879 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=3 -S < %s | FileCheck %s
+
+; Every loop here fails a legality check, so the pass must leave it alone. New
+; bail-out cases belong in this file.
+
+declare void @use(i32)
+declare void @prevent_clone(i32) noduplicate
+
+; Exit values are not supported yet: %acc.next escapes through an LCSSA PHI, so
+; the partitions would have to merge it at the final exit.
+
+define i64 @exit_value(i64 %n) {
+; CHECK-LABEL: define i64 @exit_value(
+; CHECK-SAME: i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[ACC:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[ACC_NEXT]] = add i64 [[ACC]], [[IV]]
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[R:%.*]] = phi i64 [ [[ACC_NEXT]], %[[LOOP]] ]
+; CHECK-NEXT:    ret i64 [[R]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i64 [ 0, %entry ], [ %acc.next, %loop ]
+  %acc.next = add i64 %acc, %iv
+  %iv.next = add nsw i64 %iv, 1
+  %c = icmp slt i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  %r = phi i64 [ %acc.next, %loop ]
+  ret i64 %r
+}
+
+; The latch compare itself can be the escaping value. Rewriting the latch would
+; displace it while it still has a use outside the loop.
+
+define i1 @latchcmp_liveout(i32 %n) {
+; CHECK-LABEL: define i1 @latchcmp_liveout(
+; CHECK-SAME: i32 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    call void @use(i32 [[IV]])
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[CMP_LCSSA:%.*]] = phi i1 [ [[CMP]], %[[LOOP]] ]
+; CHECK-NEXT:    ret i1 [[CMP_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  call void @use(i32 %iv)
+  %iv.next = add nsw i32 %iv, 1
+  %cmp = icmp slt i32 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  %cmp.lcssa = phi i1 [ %cmp, %loop ]
+  ret i1 %cmp.lcssa
+}
+
+; The same, with an extra in-loop use of the compare.
+
+define i1 @latchcmp_liveout_extra_use(i32 %n) {
+; CHECK-LABEL: define i1 @latchcmp_liveout_extra_use(
+; CHECK-SAME: i32 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    call void @use(i32 [[IV]])
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    [[CMP_EXT:%.*]] = zext i1 [[CMP]] to i32
+; CHECK-NEXT:    call void @use(i32 [[CMP_EXT]])
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[CMP_LCSSA:%.*]] = phi i1 [ [[CMP]], %[[LOOP]] ]
+; CHECK-NEXT:    ret i1 [[CMP_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  call void @use(i32 %iv)
+  %iv.next = add nsw i32 %iv, 1
+  %cmp = icmp slt i32 %iv.next, %n
+  %cmp.ext = zext i1 %cmp to i32
+  call void @use(i32 %cmp.ext)
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  %cmp.lcssa = phi i1 [ %cmp, %loop ]
+  ret i1 %cmp.lcssa
+}
+
+; Two exits: the exiting block is not the latch, so the loop is not bottom
+; tested.
+
+define void @multiple_exits(ptr %a, i64 %n, i1 %bail) {
+; CHECK-LABEL: define void @multiple_exits(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i1 [[BAIL:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    br i1 [[BAIL]], label %[[EXIT2:.*]], label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+; CHECK:       [[EXIT2]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  br i1 %bail, label %exit2, label %latch
+
+latch:
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nsw i64 %iv, 1
+  %c = icmp slt i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+
+exit2:
+  ret void
+}
+
+; Only a unit step is handled so far.
+
+define void @non_unit_step(ptr %a, i64 %n) {
+; CHECK-LABEL: define void @non_unit_step(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 2
+; CHECK-NEXT:    [[C:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nsw i64 %iv, 2
+  %c = icmp slt i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The trip count must be computable: the exit test here depends on a load.
+
+define void @uncomputable_trip_count(ptr %a, ptr %q) {
+; CHECK-LABEL: define void @uncomputable_trip_count(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[Q:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[Q]], align 4
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[IV]]
+; CHECK-NEXT:    store i32 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp ne i32 [[V]], 0
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %v = load i32, ptr %q
+  %p = getelementptr inbounds i32, ptr %a, i32 %iv
+  store i32 %iv, ptr %p
+  %iv.next = add nsw i32 %iv, 1
+  %c = icmp ne i32 %v, 0
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; An eq/ne latch predicate carries no ordering, and without a no-wrap flag on the
+; recurrence there is nothing else to infer it from, so the guard and latch
+; predicates cannot be given a signedness.
+
+define void @unknown_signedness(ptr %a, i64 %n, i64 %s) {
+; CHECK-LABEL: define void @unknown_signedness(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i64 [[S:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[S]], %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %s, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add i64 %iv, 1
+  %c = icmp ne i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The latch compare is rewritten in place, so it has to live in the latch. Here
+; it sits in a block that merely dominates the latch.
+
+define void @latchcmp_outside_latch(ptr %a, i64 %n) {
+; CHECK-LABEL: define void @latchcmp_outside_latch(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %latch ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nsw i64 %iv, 1
+  %c = icmp slt i64 %iv.next, %n
+  br label %latch
+
+latch:
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; Loop-carried values are not supported yet: partition 1 would have to resume
+; %acc from wherever partition 0 left it, which means rebuilding SSA. The split
+; is purely structural, so any header PHI other than the induction is refused.
+
+define void @carried_accumulator(ptr %a, i64 %n) {
+; CHECK-LABEL: define void @carried_accumulator(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[ACC:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[ACC_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[ACC]], ptr [[P]], align 4
+; CHECK-NEXT:    [[ACC_NEXT]] = add i64 [[ACC]], 3
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %acc = phi i64 [ 0, %entry ], [ %acc.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %acc, ptr %p
+  %acc.next = add i64 %acc, 3
+  %iv.next = add nsw i64 %iv, 1
+  %c = icmp slt i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The same for a secondary pointer induction, which is carried just as much as
+; an accumulator is.
+
+define void @carried_pointer_iv(ptr %a, i64 %n) {
+; CHECK-LABEL: define void @carried_pointer_iv(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = phi ptr [ [[A]], %[[ENTRY]] ], [ [[P_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[P_NEXT]] = getelementptr inbounds i64, ptr [[P]], i64 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = phi ptr [ %a, %entry ], [ %p.next, %loop ]
+  store i64 %iv, ptr %p
+  %p.next = getelementptr inbounds i64, ptr %p, i64 1
+  %iv.next = add nsw i64 %iv, 1
+  %c = icmp slt i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; An LCSSA PHI whose incoming value is a constant rather than an instruction: no
+; instruction in the loop has a use outside it, so only the PHI itself gives the
+; escaping value away.
+
+define i32 @exit_value_from_constant() {
+; CHECK-LABEL: define i32 @exit_value_from_constant() {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
+; CHECK-NEXT:    br label %[[INNER:.*]]
+; CHECK:       [[INNER]]:
+; CHECK-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[OUTER_HEADER]] ], [ [[J_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT:    [[J_NEXT]] = add nuw nsw i64 [[J]], 1
+; CHECK-NEXT:    [[EC_J:%.*]] = icmp slt i64 [[J_NEXT]], 2
+; CHECK-NEXT:    br i1 [[EC_J]], label %[[INNER]], label %[[INNER_EXIT:.*]]
+; CHECK:       [[INNER_EXIT]]:
+; CHECK-NEXT:    [[CST:%.*]] = phi i32 [ 42, %[[INNER]] ]
+; CHECK-NEXT:    br label %[[OUTER_LATCH]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 1
+; CHECK-NEXT:    [[EC_I:%.*]] = icmp slt i64 [[I_NEXT]], 2
+; CHECK-NEXT:    br i1 [[EC_I]], label %[[OUTER_HEADER]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[CST_LCSSA:%.*]] = phi i32 [ [[CST]], %[[OUTER_LATCH]] ]
+; CHECK-NEXT:    ret i32 [[CST_LCSSA]]
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %outer.latch ]
+  br label %inner
+
+inner:
+  %j = phi i64 [ 0, %outer.header ], [ %j.next, %inner ]
+  %j.next = add nuw nsw i64 %j, 1
+  %ec.j = icmp slt i64 %j.next, 2
+  br i1 %ec.j, label %inner, label %inner.exit
+
+inner.exit:
+  %cst = phi i32 [ 42, %inner ]
+  br label %outer.latch
+
+outer.latch:
+  %i.next = add nuw nsw i64 %i, 1
+  %ec.i = icmp slt i64 %i.next, 2
+  br i1 %ec.i, label %outer.header, label %exit
+
+exit:
+  %cst.lcssa = phi i32 [ %cst, %outer.latch ]
+  ret i32 %cst.lcssa
+}
+
+; A token-like target extension type cannot appear in a PHI, so LCSSA leaves
+; %resource escaping the loop with no exit PHI to find. Legality has to look for
+; the outside use itself, or the store below stops being dominated by its
+; operand once the exit block is split.
+
+define void @token_like_exit_value(ptr %resource.ptr) {
+; CHECK-LABEL: define void @token_like_exit_value(
+; CHECK-SAME: ptr [[RESOURCE_PTR:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[RESOURCE:%.*]] = load target("dx.RawBuffer", i32, 1, 0), ptr [[RESOURCE_PTR]], align 8
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp slt i32 [[IV_NEXT]], 100
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    store target("dx.RawBuffer", i32, 1, 0) [[RESOURCE]], ptr [[RESOURCE_PTR]], align 8
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %resource = load target("dx.RawBuffer", i32, 1, 0), ptr %resource.ptr
+  %iv.next = add nsw i32 %iv, 1
+  %c = icmp slt i32 %iv.next, 100
+  br i1 %c, label %loop, label %exit
+
+exit:
+  store target("dx.RawBuffer", i32, 1, 0) %resource, ptr %resource.ptr
+  ret void
+}
+
+; Florian's reproducer: the second loop's carried value starts from the first
+; loop's exit value. Both loops have exit values, so both are rejected and the
+; function is unchanged.
+
+define i64 @chained_reductions(ptr %a, i64 %n) {
+; CHECK-LABEL: define i64 @chained_reductions(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP1:.*]]
+; CHECK:       [[LOOP1]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT:    [[S:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[S_NEXT:%.*]], %[[LOOP1]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[I]]
+; CHECK-NEXT:    [[V:%.*]] = load i64, ptr [[P]], align 4
+; CHECK-NEXT:    [[S_NEXT]] = add i64 [[S]], [[V]]
+; CHECK-NEXT:    [[I_NEXT]] = add nuw nsw i64 [[I]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp slt i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP1]], label %[[MID:.*]]
+; CHECK:       [[MID]]:
+; CHECK-NEXT:    [[S1:%.*]] = phi i64 [ [[S_NEXT]], %[[LOOP1]] ]
+; CHECK-NEXT:    br label %[[LOOP2:.*]]
+; CHECK:       [[LOOP2]]:
+; CHECK-NEXT:    [[J:%.*]] = phi i64 [ 0, %[[MID]] ], [ [[J_NEXT:%.*]], %[[LOOP2]] ]
+; CHECK-NEXT:    [[T:%.*]] = phi i64 [ [[S1]], %[[MID]] ], [ [[T_NEXT:%.*]], %[[LOOP2]] ]
+; CHECK-NEXT:    [[Q:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[J]]
+; CHECK-NEXT:    [[W:%.*]] = load i64, ptr [[Q]], align 4
+; CHECK-NEXT:    [[T_NEXT]] = sub i64 [[T]], [[W]]
+; CHECK-NEXT:    [[J_NEXT]] = add nuw nsw i64 [[J]], 1
+; CHECK-NEXT:    [[D:%.*]] = icmp slt i64 [[J_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[D]], label %[[LOOP2]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[R:%.*]] = phi i64 [ [[T_NEXT]], %[[LOOP2]] ]
+; CHECK-NEXT:    ret i64 [[R]]
+;
+entry:
+  br label %loop1
+
+loop1:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop1 ]
+  %s = phi i64 [ 0, %entry ], [ %s.next, %loop1 ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %i
+  %v = load i64, ptr %p
+  %s.next = add i64 %s, %v
+  %i.next = add nuw nsw i64 %i, 1
+  %c = icmp slt i64 %i.next, %n
+  br i1 %c, label %loop1, label %mid
+
+mid:
+  %s1 = phi i64 [ %s.next, %loop1 ]
+  br label %loop2
+
+loop2:
+  %j = phi i64 [ 0, %mid ], [ %j.next, %loop2 ]
+  %t = phi i64 [ %s1, %mid ], [ %t.next, %loop2 ]
+  %q = getelementptr inbounds i64, ptr %a, i64 %j
+  %w = load i64, ptr %q
+  %t.next = sub i64 %t, %w
+  %j.next = add nuw nsw i64 %j, 1
+  %d = icmp slt i64 %j.next, %n
+  br i1 %d, label %loop2, label %exit
+
+exit:
+  %r = phi i64 [ %t.next, %loop2 ]
+  ret i64 %r
+}
+
+; Splitting a loop clones it, so instructions that must not be duplicated block
+; the transform. Both loops below are otherwise perfectly splittable.
+
+define void @cannot_clone_noduplicate_call(i32 %n) {
+; CHECK-LABEL: define void @cannot_clone_noduplicate_call(
+; CHECK-SAME: i32 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    call void @prevent_clone(i32 [[IV]])
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  call void @prevent_clone(i32 %iv)
+  %iv.next = add nsw i32 %iv, 1
+  %cmp = icmp slt i32 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+define void @cannot_clone_indirectbr(i32 %n, i1 %which) {
+; CHECK-LABEL: define void @cannot_clone_indirectbr(
+; CHECK-SAME: i32 [[N:%.*]], i1 [[WHICH:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[BR_ADDR:%.*]] = select i1 [[WHICH]], ptr blockaddress(@cannot_clone_indirectbr, %[[BLOCK_A:.*]]), ptr blockaddress(@cannot_clone_indirectbr, %[[BLOCK_B:.*]])
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    indirectbr ptr [[BR_ADDR]], [label %[[BLOCK_A]], label %[[BLOCK_B]]]
+; CHECK:       [[BLOCK_A]]:
+; CHECK-NEXT:    [[VAL_A:%.*]] = add i32 [[IV]], 100
+; CHECK-NEXT:    br label %[[LATCH]]
+; CHECK:       [[BLOCK_B]]:
+; CHECK-NEXT:    [[VAL_B:%.*]] = add i32 [[IV]], 200
+; CHECK-NEXT:    br label %[[LATCH]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[VAL:%.*]] = phi i32 [ [[VAL_A]], %[[BLOCK_A]] ], [ [[VAL_B]], %[[BLOCK_B]] ]
+; CHECK-NEXT:    call void @use(i32 [[VAL]])
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %br.addr = select i1 %which, ptr blockaddress(@cannot_clone_indirectbr, %block.a), ptr blockaddress(@cannot_clone_indirectbr, %block.b)
+  br label %loop
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %latch ]
+  indirectbr ptr %br.addr, [label %block.a, label %block.b]
+
+block.a:
+  %val.a = add i32 %iv, 100
+  br label %latch
+
+block.b:
+  %val.b = add i32 %iv, 200
+  br label %latch
+
+latch:
+  %val = phi i32 [ %val.a, %block.a ], [ %val.b, %block.b ]
+  call void @use(i32 %val)
+  %iv.next = add nsw i32 %iv, 1
+  %cmp = icmp slt i32 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; A loop whose only induction is a pointer: the partition bounds are integer
+; arithmetic on the induction type, so there is nothing to compute them in.
+
+define void @pointer_induction(ptr %src, i64 %offset) {
+; CHECK-LABEL: define void @pointer_induction(
+; CHECK-SAME: ptr [[SRC:%.*]], i64 [[OFFSET:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[END:%.*]] = getelementptr i8, ptr [[SRC]], i64 [[OFFSET]]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[PTR_IV:%.*]] = phi ptr [ [[SRC]], %[[ENTRY]] ], [ [[PTR_IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    store i8 0, ptr [[PTR_IV]], align 1
+; CHECK-NEXT:    [[PTR_IV_NEXT]] = getelementptr nusw i8, ptr [[PTR_IV]], i64 1
+; CHECK-NEXT:    [[C:%.*]] = icmp ne ptr [[PTR_IV_NEXT]], [[END]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %end = getelementptr i8, ptr %src, i64 %offset
+  br label %loop
+
+loop:
+  %ptr.iv = phi ptr [ %src, %entry ], [ %ptr.iv.next, %loop ]
+  store i8 0, ptr %ptr.iv
+  %ptr.iv.next = getelementptr nusw i8, ptr %ptr.iv, i64 1
+  %c = icmp ne ptr %ptr.iv.next, %end
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The exit block is shared with a path that bypasses the loop, so the loop does
+; not have dedicated exits.
+
+define void @non_dedicated_exit(ptr %a, i64 %n, i1 %skip) {
+; CHECK-LABEL: define void @non_dedicated_exit(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i1 [[SKIP:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br i1 [[SKIP]], label %[[EXIT:.*]], label %[[PH:.*]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp slt i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br i1 %skip, label %exit, label %ph
+
+ph:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %ph ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nsw i64 %iv, 1
+  %c = icmp slt i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; Partition bounds and entry guards assume the space runs monotonically from
+; start to end. Where the induction can wrap past the type extreme that
+; assumption breaks and a sub-loop can iterate forever, so the loop is rejected.
+; wrapping-iteration-space.ll covers the starts that are accepted.
+
+; Nothing rules out %start being above %n, in which case the induction wraps all
+; the way around the i8 range.
+
+define void @wrapping_symbolic_start(ptr %a, i8 %start, i8 %n) {
+; CHECK-LABEL: define void @wrapping_symbolic_start(
+; CHECK-SAME: ptr [[A:%.*]], i8 [[START:%.*]], i8 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ [[START]], %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i8 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp ult i8 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ %start, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, 1
+  %c = icmp ult i8 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; The same shape counting down: %start may be below %n.
+
+define void @wrapping_symbolic_start_descending(ptr %a, i8 %start, i8 %n) {
+; CHECK-LABEL: define void @wrapping_symbolic_start_descending(
+; CHECK-SAME: ptr [[A:%.*]], i8 [[START:%.*]], i8 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ [[START]], %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i8 [[IV]], -1
+; CHECK-NEXT:    [[C:%.*]] = icmp ugt i8 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ %start, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, -1
+  %c = icmp ugt i8 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; Splitting into N partitions needs N-1 boundaries, and a partition that ends up
+; empty puts one just outside the space, at the step beyond the induction start.
+; Each loop below starts at the extreme of its iteration direction, where that
+; step wraps to the far end of the type and still compares as in range, so the
+; empty partition would come out covering the whole type. All four run a single
+; iteration, so there is nothing to divide either way.
+
+; Ascending with a signed ordering, starting at SMAX: `127 + 1` wraps to -128.
+
+define void @ascending_signed_start_at_smax(ptr %a) {
+; CHECK-LABEL: define void @ascending_signed_start_at_smax(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ 127, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i8 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp slt i8 [[IV]], 127
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ 127, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, 1
+  %c = icmp slt i8 %iv, 127
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; Ascending with an unsigned ordering, starting at UMAX: `255 + 1` wraps to 0.
+
+define void @ascending_unsigned_start_at_umax(ptr %a) {
+; CHECK-LABEL: define void @ascending_unsigned_start_at_umax(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ -1, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i8 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp ult i8 [[IV]], -1
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ -1, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, 1
+  %c = icmp ult i8 %iv, -1
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; Descending with a signed ordering, starting at SMIN: `-128 - 1` wraps to 127.
+
+define void @descending_signed_start_at_smin(ptr %a) {
+; CHECK-LABEL: define void @descending_signed_start_at_smin(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ -128, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i8 [[IV]], -1
+; CHECK-NEXT:    [[C:%.*]] = icmp sgt i8 [[IV]], -128
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ -128, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, -1
+  %c = icmp sgt i8 %iv, -128
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; Descending with an unsigned ordering, starting at zero: `0 - 1` wraps to 255.
+
+define void @descending_unsigned_start_at_zero(ptr %a) {
+; CHECK-LABEL: define void @descending_unsigned_start_at_zero(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i8 [[IV]], -1
+; CHECK-NEXT:    [[C:%.*]] = icmp ugt i8 [[IV]], 0
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, -1
+  %c = icmp ugt i8 %iv, 0
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
diff --git a/llvm/test/Transforms/LoopSplit/wrapping-iteration-space.ll b/llvm/test/Transforms/LoopSplit/wrapping-iteration-space.ll
new file mode 100644
index 0000000000000..e4295649f6659
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/wrapping-iteration-space.ll
@@ -0,0 +1,139 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=3 -S < %s | FileCheck %s
+
+; Partition bounds and entry guards assume the space runs monotonically from
+; start to end. A symbolic start is accepted once something rules out wrapping
+; past the type extreme; unsupported-forms.ll covers the starts that are
+; rejected.
+
+; A no-wrap flag on the recurrence asserts monotonicity directly, so a symbolic
+; start is accepted once the induction is known not to wrap.
+
+define void @nowrap_flag_still_splits(ptr %a, i8 %start, i8 %n) {
+; CHECK-LABEL: define void @nowrap_flag_still_splits(
+; CHECK-SAME: ptr [[A:%.*]], i8 [[START:%.*]], i8 [[N:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    [[TMP0:%.*]] = add nuw i8 [[START]], 1
+; CHECK-NEXT:    [[UMAX:%.*]] = call i8 @llvm.umax.i8(i8 [[N]], i8 [[TMP0]])
+; CHECK-NEXT:    [[TMP1:%.*]] = add i8 [[UMAX]], -1
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i8 [[TMP1]], [[START]]
+; CHECK-NEXT:    [[UMIN:%.*]] = call i8 @llvm.umin.i8(i8 [[TMP2]], i8 3)
+; CHECK-NEXT:    [[UMAX1:%.*]] = call i8 @llvm.umax.i8(i8 [[UMIN]], i8 1)
+; CHECK-NEXT:    [[TMP3:%.*]] = add i8 [[START]], [[UMAX1]]
+; CHECK-NEXT:    [[TMP4:%.*]] = add i8 [[TMP3]], -1
+; CHECK-NEXT:    [[UMIN2:%.*]] = call i8 @llvm.umin.i8(i8 [[TMP4]], i8 [[TMP1]])
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp ule i8 [[START]], [[UMIN2]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ [[START]], %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw i8 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp ult i8 [[IV]], [[UMIN2]]
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK4:%.*]] = icmp ule i8 [[TMP3]], [[TMP1]]
+; CHECK-NEXT:    br i1 [[ITR_CHK4]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i8 [ [[TMP3]], %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV_LS1]]
+; CHECK-NEXT:    store i8 [[IV_LS1]], ptr [[P_LS1]], align 1
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add nuw i8 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK5:%.*]] = icmp ult i8 [[IV_LS1]], [[TMP1]]
+; CHECK-NEXT:    br i1 [[ITR_CHK5]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ %start, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add nuw i8 %iv, 1
+  %c = icmp ult i8 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; A dominating condition that bounds the start is also enough: the check retries
+; under applyLoopGuards, so a guarded range is not rejected even with no no-wrap
+; flag on the recurrence.
+
+define void @guarded_start_still_splits(ptr %a, i8 %start) {
+; CHECK-LABEL: define void @guarded_start_still_splits(
+; CHECK-SAME: ptr [[A:%.*]], i8 [[START:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[OK:%.*]] = icmp ult i8 [[START]], 50
+; CHECK-NEXT:    br i1 [[OK]], label %[[LS_GUARD0:.*]], label %[[RET:.*]]
+; CHECK:       [[LS_GUARD0]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = sub i8 99, [[START]]
+; CHECK-NEXT:    [[UMIN:%.*]] = call i8 @llvm.umin.i8(i8 [[TMP0]], i8 3)
+; CHECK-NEXT:    [[UMAX:%.*]] = call i8 @llvm.umax.i8(i8 [[UMIN]], i8 1)
+; CHECK-NEXT:    [[TMP1:%.*]] = add i8 [[START]], [[UMAX]]
+; CHECK-NEXT:    [[TMP2:%.*]] = add i8 [[TMP1]], -1
+; CHECK-NEXT:    [[UMIN1:%.*]] = call i8 @llvm.umin.i8(i8 [[TMP2]], i8 99)
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp ule i8 [[START]], [[UMIN1]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[PH:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[PH]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i8 [ [[START]], %[[PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV]]
+; CHECK-NEXT:    store i8 [[IV]], ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i8 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK2:%.*]] = icmp ult i8 [[IV]], [[UMIN1]]
+; CHECK-NEXT:    br i1 [[ITR_CHK2]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp ule i8 [[TMP1]], 99
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[PH_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[PH_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i8 [ [[TMP1]], %[[PH_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i8, ptr [[A]], i8 [[IV_LS1]]
+; CHECK-NEXT:    store i8 [[IV_LS1]], ptr [[P_LS1]], align 1
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add i8 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK4:%.*]] = icmp ult i8 [[IV_LS1]], 99
+; CHECK-NEXT:    br i1 [[ITR_CHK4]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    br label %[[RET]]
+; CHECK:       [[RET]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  %ok = icmp ult i8 %start, 50
+  br i1 %ok, label %ph, label %ret
+
+ph:
+  br label %loop
+
+loop:
+  %iv = phi i8 [ %start, %ph ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i8 %iv
+  store i8 %iv, ptr %p
+  %iv.next = add i8 %iv, 1
+  %c = icmp ult i8 %iv.next, 100
+  br i1 %c, label %loop, label %exit
+
+exit:
+  br label %ret
+
+ret:
+  ret void
+}

>From d87d4de1fadf8a5e1f10bddc60a51f890761de7d Mon Sep 17 00:00:00 2001
From: Ashutosh Nema <ashu1212 at gmail.com>
Date: Wed, 26 Aug 2026 12:21:40 +0530
Subject: [PATCH 2/9] [Transforms][Utils] Address LoopSplit review comments

---
 llvm/lib/Transforms/Utils/LoopSplit.cpp       | 28 +++++++---
 .../Transforms/LoopSplit/unsupported-forms.ll | 54 +++++++++++++++++++
 .../Transforms/LoopSplit/wide-induction.ll    | 53 ++++++++++++++++++
 3 files changed, 129 insertions(+), 6 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopSplit/wide-induction.ll

diff --git a/llvm/lib/Transforms/Utils/LoopSplit.cpp b/llvm/lib/Transforms/Utils/LoopSplit.cpp
index b4eaf7de3fdd0..97c149fb7f5c2 100644
--- a/llvm/lib/Transforms/Utils/LoopSplit.cpp
+++ b/llvm/lib/Transforms/Utils/LoopSplit.cpp
@@ -64,6 +64,7 @@
 //===----------------------------------------------------------------------===//
 
 #include "llvm/Transforms/Utils/LoopSplit.h"
+#include "llvm/ADT/STLExtras.h"
 #include "llvm/Analysis/LoopInfo.h"
 #include "llvm/Analysis/ScalarEvolution.h"
 #include "llvm/Analysis/ScalarEvolutionExpressions.h"
@@ -145,23 +146,38 @@ static const SCEVAddRecExpr *analyzeInduction(Loop *L, ScalarEvolution *SE) {
 
   // One compare operand must be the induction, either the PHI or its step. The
   // rebuilt latch always compares the PHI, so which operand it was is not used.
-  if (LatchCmp->getOperand(0) == Induction ||
-      LatchCmp->getOperand(0) == StepInst ||
-      LatchCmp->getOperand(1) == Induction ||
-      LatchCmp->getOperand(1) == StepInst)
+  if (any_of(LatchCmp->operands(), [&](Value *Op) {
+        return Op == Induction || Op == StepInst;
+      }))
     return AR;
   return nullptr;
 }
 
 // Decide whether the iteration ordering is signed or unsigned; returns the
 // signedness, or nullopt if it cannot be proven.
-static std::optional<bool> computeSignedness(Loop *L,
+static std::optional<bool> computeSignedness(ScalarEvolution &SE, Loop *L,
                                              const SCEVAddRecExpr *IndAR) {
   ICmpInst::Predicate P = L->getLatchCmpInst()->getPredicate();
   // A relational predicate gives the ordering directly; for eq/ne fall back to
   // the recurrence's no-wrap flags.
   if (ICmpInst::isRelational(P))
     return ICmpInst::isSigned(P);
+  if (IndAR->hasNoSignedWrap() && IndAR->hasNoUnsignedWrap()) {
+    const ConstantRange UR = SE.getUnsignedRange(IndAR);
+    const ConstantRange SR = SE.getSignedRange(IndAR);
+    if (UR.isFullSet() && SR.isFullSet()) {
+      LLVM_DEBUG(dbgs() << DEBUG_TYPE
+                 ": ambiguous iteration ordering with both nsw and nuw\n");
+      return std::nullopt;
+    }
+    if (!SR.isFullSet())
+      return true;
+    if (!UR.isFullSet())
+      return false;
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE
+               ": cannot prove iteration ordering signedness\n");
+    return std::nullopt;
+  }
   if (IndAR->hasNoSignedWrap())
     return true;
   if (IndAR->hasNoUnsignedWrap())
@@ -247,7 +263,7 @@ bool LoopSplit::isLegal() {
       return false;
     }
 
-  std::optional<bool> Signed = computeSignedness(L, IndAR);
+  std::optional<bool> Signed = computeSignedness(*SE, L, IndAR);
   if (!Signed)
     return false;
   InductionIsSigned = *Signed;
diff --git a/llvm/test/Transforms/LoopSplit/unsupported-forms.ll b/llvm/test/Transforms/LoopSplit/unsupported-forms.ll
index 57351abae931c..d94b456c59f50 100644
--- a/llvm/test/Transforms/LoopSplit/unsupported-forms.ll
+++ b/llvm/test/Transforms/LoopSplit/unsupported-forms.ll
@@ -254,6 +254,60 @@ exit:
   ret void
 }
 
+; An eq/ne latch predicate with both nsw and nuw on the step leaves signedness
+; ambiguous, so the pass cannot pick guard and latch ordering.
+
+define void @ambiguous_signedness(ptr %a, i64 %n, i64 %s) {
+; CHECK-LABEL: define void @ambiguous_signedness(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]], i64 [[S:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ [[S]], %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[C:%.*]] = icmp ne i64 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ %s, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nuw nsw i64 %iv, 1
+  %c = icmp ne i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}
+
+; rewriteLatch expects a conditional latch terminator. An unconditional backedge
+; has no latch compare, so isLegal() rejects the loop before splitting.
+
+define void @uncond_latch() {
+; CHECK-LABEL: define void @uncond_latch() {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[IV_NEXT]] = add i64 [[IV]], 1
+; CHECK-NEXT:    br label %[[LOOP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %iv.next = add i64 %iv, 1
+  br label %loop
+}
+
 ; The latch compare is rewritten in place, so it has to live in the latch. Here
 ; it sits in a block that merely dominates the latch.
 
diff --git a/llvm/test/Transforms/LoopSplit/wide-induction.ll b/llvm/test/Transforms/LoopSplit/wide-induction.ll
new file mode 100644
index 0000000000000..809360eb07a49
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/wide-induction.ll
@@ -0,0 +1,53 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-split -loop-split-points=50 -S < %s | FileCheck %s
+
+; Wide induction types take the BitWidth >= 32 path in the test pass when
+; filtering CLI split offsets. Offsets within the unsigned CLI range are
+; accepted and split normally.
+
+define void @i128_induction(ptr %a) {
+; CHECK-LABEL: define void @i128_induction(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i128 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i8, ptr [[A]], i128 [[IV]]
+; CHECK-NEXT:    store i8 0, ptr [[P]], align 1
+; CHECK-NEXT:    [[IV_NEXT]] = add i128 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp ult i128 [[IV]], 49
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    br i1 true, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i128 [ 50, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i8, ptr [[A]], i128 [[IV_LS1]]
+; CHECK-NEXT:    store i8 0, ptr [[P_LS1]], align 1
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add i128 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp ult i128 [[IV_LS1]], 99
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i128 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i8, ptr %a, i128 %iv
+  store i8 0, ptr %p
+  %iv.next = add i128 %iv, 1
+  %c = icmp ult i128 %iv.next, 100
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}

>From 26131ac830b7e6ddb05d6f40d260b6458e4ce8a7 Mon Sep 17 00:00:00 2001
From: Ashutosh Nema <ashu1212 at gmail.com>
Date: Wed, 26 Aug 2026 15:45:18 +0530
Subject: [PATCH 3/9] [Transforms][Utils] Address LoopSplit build issue

---
 .../llvm/Transforms/Utils/LoopSplitPass.h     |  2 +-
 llvm/lib/Transforms/Utils/LoopSplit.cpp       |  5 ++--
 llvm/test/Transforms/LoopSplit/basic.ll       | 19 ++++++------
 .../Transforms/LoopSplit/branch-weights.ll    |  4 +--
 llvm/test/Transforms/LoopSplit/loop-depth.ll  | 28 ++++++++---------
 .../LoopSplit/multiple-partitions.ll          | 30 +++++++++----------
 .../Transforms/LoopSplit/saturating-latch.ll  |  4 +--
 .../Transforms/LoopSplit/sequential-loops.ll  |  8 ++---
 8 files changed, 49 insertions(+), 51 deletions(-)

diff --git a/llvm/include/llvm/Transforms/Utils/LoopSplitPass.h b/llvm/include/llvm/Transforms/Utils/LoopSplitPass.h
index 14dacfa8b5b7f..e2efafee0a74c 100644
--- a/llvm/include/llvm/Transforms/Utils/LoopSplitPass.h
+++ b/llvm/include/llvm/Transforms/Utils/LoopSplitPass.h
@@ -20,7 +20,7 @@
 
 namespace llvm {
 
-class LoopSplitPass : public PassInfoMixin<LoopSplitPass> {
+class LoopSplitPass : public OptionalPassInfoMixin<LoopSplitPass> {
 public:
   LLVM_ABI PreservedAnalyses run(Function &F, FunctionAnalysisManager &AM);
 };
diff --git a/llvm/lib/Transforms/Utils/LoopSplit.cpp b/llvm/lib/Transforms/Utils/LoopSplit.cpp
index 97c149fb7f5c2..b34662fe00cae 100644
--- a/llvm/lib/Transforms/Utils/LoopSplit.cpp
+++ b/llvm/lib/Transforms/Utils/LoopSplit.cpp
@@ -146,9 +146,8 @@ static const SCEVAddRecExpr *analyzeInduction(Loop *L, ScalarEvolution *SE) {
 
   // One compare operand must be the induction, either the PHI or its step. The
   // rebuilt latch always compares the PHI, so which operand it was is not used.
-  if (any_of(LatchCmp->operands(), [&](Value *Op) {
-        return Op == Induction || Op == StepInst;
-      }))
+  if (any_of(LatchCmp->operands(),
+             [&](Value *Op) { return Op == Induction || Op == StepInst; }))
     return AR;
   return nullptr;
 }
diff --git a/llvm/test/Transforms/LoopSplit/basic.ll b/llvm/test/Transforms/LoopSplit/basic.ll
index df513a5a37286..78555626e05fd 100644
--- a/llvm/test/Transforms/LoopSplit/basic.ll
+++ b/llvm/test/Transforms/LoopSplit/basic.ll
@@ -12,9 +12,9 @@ define void @split_signed(ptr %a, i64 %n) {
 ; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 50)
-; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
 ; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
 ; CHECK:       [[ENTRY]]:
@@ -67,12 +67,11 @@ define void @split_unsigned(ptr %a, i64 %n) {
 ; CHECK-LABEL: define void @split_unsigned(
 ; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
 ; CHECK-NEXT:  [[LS_GUARD0:.*:]]
-; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
-; CHECK-NEXT:    [[TMP0:%.*]] = add i64 [[UMAX]], -1
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[N]], i64 1)
 ; CHECK-NEXT:    [[UMIN1:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 50)
-; CHECK-NEXT:    [[UMAX1:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN1]], i64 1)
-; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX1]], -1
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN1]], i64 1)
 ; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[UMAX1:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN1]], i64 1)
 ; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp ule i64 0, [[UMIN]]
 ; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
 ; CHECK:       [[ENTRY]]:
@@ -127,9 +126,9 @@ define void @latch_compares_phi(ptr %a, i64 %m) {
 ; CHECK-NEXT:  [[LS_GUARD0:.*:]]
 ; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[M]], i64 0)
 ; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[SMAX]], i64 50)
-; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[UMAX]], -1
+; CHECK-NEXT:    [[TMP0:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[SMAX]], i64 [[TMP0]])
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
 ; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
 ; CHECK:       [[ENTRY]]:
@@ -185,9 +184,9 @@ define void @nested_inner(ptr %a, i64 %n, i64 %m) {
 ; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[M]], i64 1)
 ; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 50)
-; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
 ; CHECK:       [[OUTER_HEADER]]:
 ; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
diff --git a/llvm/test/Transforms/LoopSplit/branch-weights.ll b/llvm/test/Transforms/LoopSplit/branch-weights.ll
index 4bdd4e37f811d..e926cf0bc0743 100644
--- a/llvm/test/Transforms/LoopSplit/branch-weights.ll
+++ b/llvm/test/Transforms/LoopSplit/branch-weights.ll
@@ -13,9 +13,9 @@ define void @basic(ptr %a, i64 %n) !prof !0 {
 ; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 50)
-; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
 ; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]], !prof [[PROF1:![0-9]+]]
 ; CHECK:       [[ENTRY]]:
diff --git a/llvm/test/Transforms/LoopSplit/loop-depth.ll b/llvm/test/Transforms/LoopSplit/loop-depth.ll
index 5b4d0680eec8f..b16984877b25b 100644
--- a/llvm/test/Transforms/LoopSplit/loop-depth.ll
+++ b/llvm/test/Transforms/LoopSplit/loop-depth.ll
@@ -17,9 +17,9 @@ define void @nest2(ptr %a, i64 %n, i64 %m) {
 ; INNERMOST-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[M]], i64 1)
 ; INNERMOST-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; INNERMOST-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
-; INNERMOST-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; INNERMOST-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; INNERMOST-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; INNERMOST-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; INNERMOST-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; INNERMOST-NEXT:    br label %[[OUTER_HEADER:.*]]
 ; INNERMOST:       [[OUTER_HEADER]]:
 ; INNERMOST-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
@@ -69,9 +69,9 @@ define void @nest2(ptr %a, i64 %n, i64 %m) {
 ; DEPTH1-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; DEPTH1-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; DEPTH1-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
-; DEPTH1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; DEPTH1-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; DEPTH1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; DEPTH1-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; DEPTH1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; DEPTH1-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
 ; DEPTH1-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
 ; DEPTH1:       [[ENTRY]]:
@@ -132,9 +132,9 @@ define void @nest2(ptr %a, i64 %n, i64 %m) {
 ; DEPTH2-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[M]], i64 1)
 ; DEPTH2-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; DEPTH2-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
-; DEPTH2-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; DEPTH2-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; DEPTH2-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; DEPTH2-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; DEPTH2-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; DEPTH2-NEXT:    br label %[[OUTER_HEADER:.*]]
 ; DEPTH2:       [[OUTER_HEADER]]:
 ; DEPTH2-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[OUTER_LATCH:.*]] ]
@@ -219,9 +219,9 @@ define void @nest3(ptr %a, i64 %n) {
 ; INNERMOST-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; INNERMOST-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; INNERMOST-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
-; INNERMOST-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; INNERMOST-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; INNERMOST-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; INNERMOST-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; INNERMOST-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; INNERMOST-NEXT:    br label %[[O_HEADER:.*]]
 ; INNERMOST:       [[O_HEADER]]:
 ; INNERMOST-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[O_LATCH:.*]] ]
@@ -282,9 +282,9 @@ define void @nest3(ptr %a, i64 %n) {
 ; DEPTH1-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; DEPTH1-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; DEPTH1-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
-; DEPTH1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; DEPTH1-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; DEPTH1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; DEPTH1-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; DEPTH1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; DEPTH1-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
 ; DEPTH1-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
 ; DEPTH1:       [[ENTRY]]:
@@ -367,9 +367,9 @@ define void @nest3(ptr %a, i64 %n) {
 ; DEPTH2-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; DEPTH2-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; DEPTH2-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
-; DEPTH2-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; DEPTH2-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; DEPTH2-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; DEPTH2-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; DEPTH2-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; DEPTH2-NEXT:    br label %[[O_HEADER:.*]]
 ; DEPTH2:       [[O_HEADER]]:
 ; DEPTH2-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[O_LATCH:.*]] ]
@@ -519,9 +519,9 @@ define void @inner_carried_outer_clean(ptr %a, i64 %n, i64 %m) {
 ; DEPTH1-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; DEPTH1-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; DEPTH1-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 2)
-; DEPTH1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; DEPTH1-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; DEPTH1-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; DEPTH1-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; DEPTH1-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; DEPTH1-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
 ; DEPTH1-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
 ; DEPTH1:       [[ENTRY]]:
diff --git a/llvm/test/Transforms/LoopSplit/multiple-partitions.ll b/llvm/test/Transforms/LoopSplit/multiple-partitions.ll
index 21d068181d040..85ef7f2e8986c 100644
--- a/llvm/test/Transforms/LoopSplit/multiple-partitions.ll
+++ b/llvm/test/Transforms/LoopSplit/multiple-partitions.ll
@@ -13,13 +13,13 @@ define void @tile(ptr %a, i64 %n) {
 ; THREE-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; THREE-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; THREE-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 10)
-; THREE-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; THREE-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; THREE-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; THREE-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; THREE-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; THREE-NEXT:    [[UMIN1:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 20)
-; THREE-NEXT:    [[UMAX2:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN1]], i64 1)
-; THREE-NEXT:    [[TMP2:%.*]] = add nsw i64 [[UMAX2]], -1
+; THREE-NEXT:    [[TMP2:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN1]], i64 1)
 ; THREE-NEXT:    [[SMIN1:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP2]])
+; THREE-NEXT:    [[UMAX2:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN1]], i64 1)
 ; THREE-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
 ; THREE-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
 ; THREE:       [[ENTRY]]:
@@ -70,17 +70,17 @@ define void @tile(ptr %a, i64 %n) {
 ; FOUR-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; FOUR-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; FOUR-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 10)
-; FOUR-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; FOUR-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; FOUR-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; FOUR-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
-; FOUR-NEXT:    [[UMIN1:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 20)
-; FOUR-NEXT:    [[UMAX2:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN1]], i64 1)
-; FOUR-NEXT:    [[TMP2:%.*]] = add nsw i64 [[UMAX2]], -1
+; FOUR-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; FOUR-NEXT:    [[UMIN4:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 20)
+; FOUR-NEXT:    [[TMP2:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN4]], i64 1)
 ; FOUR-NEXT:    [[SMIN1:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP2]])
-; FOUR-NEXT:    [[UMIN4:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 30)
 ; FOUR-NEXT:    [[UMAX5:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN4]], i64 1)
-; FOUR-NEXT:    [[TMP3:%.*]] = add nsw i64 [[UMAX5]], -1
+; FOUR-NEXT:    [[UMIN5:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 30)
+; FOUR-NEXT:    [[TMP3:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN5]], i64 1)
 ; FOUR-NEXT:    [[SMIN2:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP3]])
+; FOUR-NEXT:    [[UMAX6:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN5]], i64 1)
 ; FOUR-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
 ; FOUR-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
 ; FOUR:       [[ENTRY]]:
@@ -109,12 +109,12 @@ define void @tile(ptr %a, i64 %n) {
 ; FOUR:       [[LS_EXIT1]]:
 ; FOUR-NEXT:    br label %[[LS_GUARD2]]
 ; FOUR:       [[LS_GUARD2]]:
-; FOUR-NEXT:    [[ITR_CHK6:%.*]] = icmp sle i64 [[UMAX2]], [[SMIN2]]
+; FOUR-NEXT:    [[ITR_CHK6:%.*]] = icmp sle i64 [[UMAX5]], [[SMIN2]]
 ; FOUR-NEXT:    br i1 [[ITR_CHK6]], label %[[ENTRY_LS2:.*]], label %[[LS_GUARD3:.*]]
 ; FOUR:       [[ENTRY_LS2]]:
 ; FOUR-NEXT:    br label %[[LOOP_LS2:.*]]
 ; FOUR:       [[LOOP_LS2]]:
-; FOUR-NEXT:    [[IV_LS2:%.*]] = phi i64 [ [[UMAX2]], %[[ENTRY_LS2]] ], [ [[IV_NEXT_LS2:%.*]], %[[LOOP_LS2]] ]
+; FOUR-NEXT:    [[IV_LS2:%.*]] = phi i64 [ [[UMAX5]], %[[ENTRY_LS2]] ], [ [[IV_NEXT_LS2:%.*]], %[[LOOP_LS2]] ]
 ; FOUR-NEXT:    [[P_LS2:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS2]]
 ; FOUR-NEXT:    store i64 [[IV_LS2]], ptr [[P_LS2]], align 4
 ; FOUR-NEXT:    [[IV_NEXT_LS2]] = add nsw i64 [[IV_LS2]], 1
@@ -123,12 +123,12 @@ define void @tile(ptr %a, i64 %n) {
 ; FOUR:       [[LS_EXIT2]]:
 ; FOUR-NEXT:    br label %[[LS_GUARD3]]
 ; FOUR:       [[LS_GUARD3]]:
-; FOUR-NEXT:    [[ITR_CHK8:%.*]] = icmp sle i64 [[UMAX5]], [[TMP0]]
+; FOUR-NEXT:    [[ITR_CHK8:%.*]] = icmp sle i64 [[UMAX6]], [[TMP0]]
 ; FOUR-NEXT:    br i1 [[ITR_CHK8]], label %[[ENTRY_LS3:.*]], label %[[LS_FINAL_EXIT:.*]]
 ; FOUR:       [[ENTRY_LS3]]:
 ; FOUR-NEXT:    br label %[[LOOP_LS3:.*]]
 ; FOUR:       [[LOOP_LS3]]:
-; FOUR-NEXT:    [[IV_LS3:%.*]] = phi i64 [ [[UMAX5]], %[[ENTRY_LS3]] ], [ [[IV_NEXT_LS3:%.*]], %[[LOOP_LS3]] ]
+; FOUR-NEXT:    [[IV_LS3:%.*]] = phi i64 [ [[UMAX6]], %[[ENTRY_LS3]] ], [ [[IV_NEXT_LS3:%.*]], %[[LOOP_LS3]] ]
 ; FOUR-NEXT:    [[P_LS3:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS3]]
 ; FOUR-NEXT:    store i64 [[IV_LS3]], ptr [[P_LS3]], align 4
 ; FOUR-NEXT:    [[IV_NEXT_LS3]] = add nsw i64 [[IV_LS3]], 1
diff --git a/llvm/test/Transforms/LoopSplit/saturating-latch.ll b/llvm/test/Transforms/LoopSplit/saturating-latch.ll
index 35bc5733a3c60..d0d5a17c6c8e0 100644
--- a/llvm/test/Transforms/LoopSplit/saturating-latch.ll
+++ b/llvm/test/Transforms/LoopSplit/saturating-latch.ll
@@ -173,9 +173,9 @@ define void @symbolic_end_may_reach_umax(ptr %a, i8 %n) {
 ; CHECK-NEXT:  [[LS_GUARD0:.*:]]
 ; CHECK-NEXT:    [[TMP0:%.*]] = add i8 [[N]], -1
 ; CHECK-NEXT:    [[UMIN:%.*]] = call i8 @llvm.umin.i8(i8 [[TMP0]], i8 3)
-; CHECK-NEXT:    [[UMAX:%.*]] = call i8 @llvm.umax.i8(i8 [[UMIN]], i8 1)
-; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i8 [[UMAX]], -1
+; CHECK-NEXT:    [[TMP1:%.*]] = call i8 @llvm.usub.sat.i8(i8 [[UMIN]], i8 1)
 ; CHECK-NEXT:    [[UMIN1:%.*]] = call i8 @llvm.umin.i8(i8 [[TMP0]], i8 [[TMP1]])
+; CHECK-NEXT:    [[UMAX:%.*]] = call i8 @llvm.umax.i8(i8 [[UMIN]], i8 1)
 ; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp ule i8 0, [[UMIN1]]
 ; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
 ; CHECK:       [[ENTRY]]:
diff --git a/llvm/test/Transforms/LoopSplit/sequential-loops.ll b/llvm/test/Transforms/LoopSplit/sequential-loops.ll
index 00b0122ed2fe0..570ef144db63c 100644
--- a/llvm/test/Transforms/LoopSplit/sequential-loops.ll
+++ b/llvm/test/Transforms/LoopSplit/sequential-loops.ll
@@ -13,9 +13,9 @@ define void @two_loops(ptr %a, i64 %n) {
 ; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
 ; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 50)
-; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
-; CHECK-NEXT:    [[TMP1:%.*]] = add nsw i64 [[UMAX]], -1
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
 ; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
 ; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
 ; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
 ; CHECK:       [[ENTRY]]:
@@ -47,9 +47,9 @@ define void @two_loops(ptr %a, i64 %n) {
 ; CHECK-NEXT:    [[SMAX6:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
 ; CHECK-NEXT:    [[TMP2:%.*]] = add nsw i64 [[SMAX6]], -1
 ; CHECK-NEXT:    [[UMIN7:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP2]], i64 50)
-; CHECK-NEXT:    [[UMAX8:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN7]], i64 1)
-; CHECK-NEXT:    [[TMP3:%.*]] = add nsw i64 [[UMAX8]], -1
+; CHECK-NEXT:    [[TMP3:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN7]], i64 1)
 ; CHECK-NEXT:    [[SMIN9:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP2]], i64 [[TMP3]])
+; CHECK-NEXT:    [[UMAX8:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN7]], i64 1)
 ; CHECK-NEXT:    [[ITR_CHK12:%.*]] = icmp sle i64 0, [[SMIN9]]
 ; CHECK-NEXT:    br i1 [[ITR_CHK12]], label %[[LS_FINAL_EXIT:.*]], label %[[LS_GUARD111:.*]]
 ; CHECK:       [[LS_FINAL_EXIT]]:

>From 4d492477bbf7dc2e8049290c2f460424f59c4f88 Mon Sep 17 00:00:00 2001
From: Ashutosh Nema <ashu1212 at gmail.com>
Date: Fri, 28 Aug 2026 15:19:50 +0530
Subject: [PATCH 4/9] [Transforms][Utils] LoopSplit review follow-ups

---
 .../include/llvm/Transforms/Utils/LoopSplit.h |  57 ++++----
 llvm/lib/Transforms/Utils/LoopSplit.cpp       | 127 ++++++++++--------
 llvm/lib/Transforms/Utils/LoopSplitPass.cpp   |  23 ++--
 .../Transforms/LoopSplit/unsupported-forms.ll |   2 +-
 .../LoopSplit/wrapping-iteration-space.ll     |  51 +++++++
 5 files changed, 165 insertions(+), 95 deletions(-)

diff --git a/llvm/include/llvm/Transforms/Utils/LoopSplit.h b/llvm/include/llvm/Transforms/Utils/LoopSplit.h
index b075d76b843ad..a667676b4f48f 100644
--- a/llvm/include/llvm/Transforms/Utils/LoopSplit.h
+++ b/llvm/include/llvm/Transforms/Utils/LoopSplit.h
@@ -17,6 +17,7 @@
 #include "llvm/ADT/SmallVector.h"
 #include "llvm/Analysis/LoopInfo.h"
 #include "llvm/Support/Compiler.h"
+#include <optional>
 
 namespace llvm {
 
@@ -29,33 +30,30 @@ class ScalarEvolution;
 ///
 /// Usage:
 /// \code
-///   LoopSplit LS(L, LI, SE, DT);
-///   if (!LS.isLegal())
-///     return false;
-///   LS.addPartition(S0, E0);   // one call per partition, in order
-///   LS.addPartition(S1, E1);
-///   LS.split();
+///   if (auto LS = LoopSplit::get(L, LI, SE, DT)) {
+///     LS->addPartition(S0, E0);   // one call per partition, in order
+///     LS->addPartition(S1, E1);
+///     LS->split();
+///   }
 /// \endcode
 class LoopSplit {
 public:
-  LLVM_ABI LoopSplit(Loop *L, LoopInfo *LI, ScalarEvolution *SE,
-                     DominatorTree *DT)
-      : L(L), LI(LI), SE(SE), DT(DT) {}
-
-  /// Analyze \p L and return true if it is a counted loop this utility can
-  /// split: a bottom-tested single-exit loop in LCSSA form with dedicated
-  /// exits, no loop-carried and no escaping values, a unique unit-step integer
-  /// induction, and a computable trip count that cannot wrap. Must succeed
-  /// before split().
-  LLVM_ABI bool isLegal();
-
-  /// Return the loop's induction variable. Valid only after isLegal() succeeds.
+  /// Analyze \p L and, if it is a counted loop this utility can split, return a
+  /// LoopSplit ready for addPartition() and split(). Otherwise return
+  /// std::nullopt. Eligible loops are bottom-tested single-exit loops in LCSSA
+  /// form with dedicated exits, no loop-carried and no escaping values, a
+  /// unique unit-step integer induction, and a computable trip count that
+  /// cannot wrap.
+  LLVM_ABI static std::optional<LoopSplit>
+  get(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT);
+
+  /// Return the loop's induction variable. Valid only on a legal LoopSplit.
   LLVM_ABI PHINode *getInductionVariable() const {
     return L->getInductionVariable(*SE);
   }
 
   /// The induction value on the last iteration, which the final partition must
-  /// end at. Valid only after isLegal() succeeds.
+  /// end at. Valid only on a legal LoopSplit.
   LLVM_ABI const SCEV *getInductionEnd() const { return InductionEnd; }
 
   /// Append an inclusive partition range [Start, End] in iteration order.
@@ -65,19 +63,24 @@ class LoopSplit {
   ///
   /// Both bounds must have the induction type and be loop-invariant. They must
   /// also stay within the iteration space, extended by the one step past its
-  /// start that an empty partition needs; isLegal() has proven that much
-  /// representable. Reaching further wraps past TYPE_MAX/MIN/0 into a bound
-  /// that still looks in range, which silently miscompiles. See LoopSplit.cpp
-  /// for the rationale.
+  /// start that an empty partition needs; legality analysis has proven that
+  /// much representable. Reaching further wraps past TYPE_MAX/MIN/0 into a
+  /// bound that still looks in range, which silently miscompiles. See
+  /// LoopSplit.cpp for the rationale.
   LLVM_ABI void addPartition(const SCEV *Start, const SCEV *End);
 
-  LLVM_ABI unsigned getNumPartitions() const { return Partitions.size(); }
+  LLVM_ABI size_t getNumPartitions() const { return Partitions.size(); }
 
-  /// Perform the split. Requires a successful isLegal() and at least two
-  /// partitions. Returns true if the loop was rewritten.
+  /// Perform the split. Requires at least two partitions. Returns true if the
+  /// loop was rewritten.
   LLVM_ABI bool split();
 
 private:
+  LoopSplit(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT,
+            const SCEV *InductionEnd, bool InductionIsSigned, bool Descending)
+      : L(L), LI(LI), SE(SE), DT(DT), InductionEnd(InductionEnd),
+        InductionIsSigned(InductionIsSigned), Descending(Descending) {}
+
   /// Everything known about one partition: the caller-supplied range plus the
   /// state split() derives. Indexed by partition number in \c Partitions.
   struct PartitionInfo {
@@ -108,7 +111,7 @@ class LoopSplit {
   ScalarEvolution *SE;
   DominatorTree *DT;
 
-  // Induction analysis, populated by isLegal().
+  // Induction analysis, populated during legality analysis.
   const SCEV *InductionEnd = nullptr; // value on the last iteration.
   bool InductionIsSigned = false;     // iteration ordering signedness.
   bool Descending = false;            // step is -1 (the loop counts down).
diff --git a/llvm/lib/Transforms/Utils/LoopSplit.cpp b/llvm/lib/Transforms/Utils/LoopSplit.cpp
index b34662fe00cae..ff35753403259 100644
--- a/llvm/lib/Transforms/Utils/LoopSplit.cpp
+++ b/llvm/lib/Transforms/Utils/LoopSplit.cpp
@@ -31,19 +31,20 @@
 // The latch keeps iterating while the value the next iteration would use is
 // still in the partition. That is written as the strict "i < sel_i" on the
 // induction PHI rather than "i + 1 <= sel_i" on the step value; the two agree
-// because isLegal() has established that the space does not wrap, and the
-// strict form never forms i + 1, so it remains a real test even when sel_i is
-// the last value of the type, where the inclusive one would be a tautology and
-// the partition would never exit.
+// because legality analysis has established that the space does not wrap, and
+// the strict form never forms i + 1, so it remains a real test even when sel_i
+// is the last value of the type, where the inclusive one would be a tautology
+// and the partition would never exit.
 //
 // A descending (step -1) loop uses the same structure mirrored: partitions run
 // high-to-low and the clamp and predicates flip (>=/>).
 //
 // Usage guidelines:
 //  - Caller bounds must not wrap the induction type. The clamp absorbs a bound
-//    past the runtime trip count, and isLegal() reserves the one step past the
-//    induction start that an empty partition needs, but a bound reaching any
-//    further wraps in the bound arithmetic and cannot be repaired here.
+//    past the runtime trip count, and legality analysis reserves the one step
+//    past the induction start that an empty partition needs, but a bound
+//    reaching any further wraps in the bound arithmetic and cannot be repaired
+//    here.
 //  - Bounds must be loop-invariant: they are expanded in guard0, so a bound
 //    depending on a value defined inside the loop cannot be placed.
 //  - The partitions must tile the original iteration space exactly -- same
@@ -54,12 +55,12 @@
 // rebuilds a value that flows between partitions, so no SSA reconstruction is
 // needed.
 //
-// Not yet supported, and rejected by isLegal(): loop-carried values, values
-// that escape the loop (exit values), non-unit and non-integer inductions,
-// top-tested loops, and multiple exits. Also rejected is an induction start at
-// the extreme of the iteration direction, which leaves nowhere to put a
-// boundary. An induction *end* at that extreme is fine, because the latch stays
-// strict.
+// Not yet supported, and rejected during legality analysis: loop-carried
+// values, values that escape the loop (exit values), non-unit and non-integer
+// inductions, top-tested loops, and multiple exits. Also rejected is an
+// induction start at the extreme of the iteration direction, which leaves
+// nowhere to put a boundary. An induction *end* at that extreme is fine,
+// because the latch stays strict.
 //
 //===----------------------------------------------------------------------===//
 
@@ -94,7 +95,7 @@ using namespace llvm::SCEVPatternMatch;
 
 /// Per-split() scratch shared by the phase helpers; lives for one split() call.
 /// Everything derived from the induction lives on LoopSplit itself, filled in
-/// by isLegal(); this holds only what the transform creates.
+/// by legality analysis; this holds only what the transform creates.
 struct LoopSplit::SplitState {
   // Partition 0 reuses the original loop's preheader, exit, and entry guard;
   // those blocks live in Partitions[0] rather than being duplicated here.
@@ -105,7 +106,7 @@ struct LoopSplit::SplitState {
 
 // Record a new partition with the given inclusive iteration range.
 void LoopSplit::addPartition(const SCEV *Start, const SCEV *End) {
-  assert(InductionEnd && "addPartition() requires a successful isLegal()");
+  assert(InductionEnd && "addPartition() requires prior legality analysis");
   // The bounds are combined with the induction end and expanded in its type. A
   // mismatch would otherwise surface either as a bare "Operand types don't
   // match!" from inside ScalarEvolution, or worse, as a silent cast.
@@ -197,15 +198,36 @@ static bool isEntryGuardedByCond(ScalarEvolution &SE, Loop *L,
                                      SE.applyLoopGuards(RHS, L));
 }
 
+// Latch "keep iterating" predicate, comparing the induction PHI against the
+// partition end: `i < sel` ascending, `i > sel` descending.
+static ICmpInst::Predicate continuePredicate(bool Signed, bool Descending) {
+  ICmpInst::Predicate P = Descending ? ICmpInst::ICMP_UGT : ICmpInst::ICMP_ULT;
+  return Signed ? ICmpInst::getSignedPredicate(P) : P;
+}
+
+// Guard "enter this partition" predicate: the latch test made non-strict, so
+// `start <= sel` ascending and `start >= sel` descending. Also used during
+// legality analysis to prove the iteration space is monotonic.
+static ICmpInst::Predicate guardPredicate(bool Signed, bool Descending) {
+  return ICmpInst::getNonStrictPredicate(continuePredicate(Signed, Descending));
+}
+
+struct LoopSplitAnalysis {
+  const SCEV *InductionEnd;
+  bool InductionIsSigned;
+  bool Descending;
+};
+
 // Check every structural precondition and record the induction analysis.
-bool LoopSplit::isLegal() {
+static std::optional<LoopSplitAnalysis>
+analyzeLegality(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT) {
   // Require a bottom-tested single-exit loop in LCSSA form. Simplify form gives
   // the preheader, single latch and dedicated exits; the rest pin the exit to
   // the latch, so the latch compare can be rewritten per partition.
   if (!L->isLoopSimplifyForm() || !L->isLCSSAForm(*DT) ||
       L->getExitingBlock() != L->getLoopLatch() || !L->getExitBlock()) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop not in expected form\n");
-    return false;
+    return std::nullopt;
   }
 
   // The latch compare must exist and reside in the latch: it is rewritten in
@@ -213,7 +235,7 @@ bool LoopSplit::isLegal() {
   ICmpInst *LatchCmp = L->getLatchCmpInst();
   if (!LatchCmp || LatchCmp->getParent() != L->getLoopLatch()) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE ": latch compare not in the loop latch\n");
-    return false;
+    return std::nullopt;
   }
 
   // Exit values are unsupported. Look for an LCSSA PHI and for a use outside
@@ -221,34 +243,34 @@ bool LoopSplit::isLegal() {
   // value escaping with no PHI to find.
   if (!L->getExitBlock()->phis().empty()) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop has exit values\n");
-    return false;
+    return std::nullopt;
   }
   for (BasicBlock *BB : L->blocks())
     for (Instruction &I : *BB)
       for (User *U : I.users())
         if (auto *UI = dyn_cast<Instruction>(U); UI && !L->contains(UI)) {
           LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop has exit values\n");
-          return false;
+          return std::nullopt;
         }
 
   // Splitting a loop clones it, so cloning must be safe.
   if (!L->isSafeToClone()) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop not safe to clone\n");
-    return false;
+    return std::nullopt;
   }
 
   // A computable backedge-taken count fixes the iteration space we rebuild.
   const SCEV *BTC = SE->getBackedgeTakenCount(L);
   if (isa<SCEVCouldNotCompute>(BTC)) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop trip count uncomputable\n");
-    return false;
+    return std::nullopt;
   }
 
   const SCEVAddRecExpr *IndAR = analyzeInduction(L, SE);
   if (!IndAR) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE
                ": no unique unit-step integer induction\n");
-    return false;
+    return std::nullopt;
   }
 
   PHINode *Induction = L->getInductionVariable(*SE);
@@ -259,23 +281,23 @@ bool LoopSplit::isLegal() {
   for (PHINode &HeaderPHI : L->getHeader()->phis())
     if (&HeaderPHI != Induction) {
       LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop has carried values\n");
-      return false;
+      return std::nullopt;
     }
 
   std::optional<bool> Signed = computeSignedness(*SE, L, IndAR);
   if (!Signed)
-    return false;
-  InductionIsSigned = *Signed;
-  Descending =
+    return std::nullopt;
+  const bool InductionIsSigned = *Signed;
+  const bool Descending =
       cast<SCEVConstant>(IndAR->getStepRecurrence(*SE))->getAPInt().isAllOnes();
 
   // Start and end must share the induction type; reject any width mismatch.
   // evaluateAtIteration coerces to the start's type for an affine recurrence,
   // so this is defensive rather than reachable.
-  InductionEnd = IndAR->evaluateAtIteration(BTC, *SE);
+  const SCEV *InductionEnd = IndAR->evaluateAtIteration(BTC, *SE);
   if (InductionEnd->getType() != IndAR->getStart()->getType()) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE ": induction end/start type mismatch\n");
-    return false;
+    return std::nullopt;
   }
 
   // Partition bounds and entry guards assume the space runs monotonically from
@@ -283,15 +305,12 @@ bool LoopSplit::isLegal() {
   // flag on the recurrence asserts that directly.
   bool NoWrap =
       InductionIsSigned ? IndAR->hasNoSignedWrap() : IndAR->hasNoUnsignedWrap();
-  ICmpInst::Predicate Ordered =
-      Descending
-          ? (InductionIsSigned ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE)
-          : (InductionIsSigned ? ICmpInst::ICMP_SLE : ICmpInst::ICMP_ULE);
-  if (!NoWrap &&
-      !isEntryGuardedByCond(*SE, L, Ordered, IndAR->getStart(), InductionEnd)) {
+  if (!NoWrap && !isEntryGuardedByCond(
+                     *SE, L, guardPredicate(InductionIsSigned, Descending),
+                     IndAR->getStart(), InductionEnd)) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE
                ": iteration space may wrap past the type extreme\n");
-    return false;
+    return std::nullopt;
   }
 
   // A boundary can sit one step beyond the start, so that step has to be
@@ -302,36 +321,32 @@ bool LoopSplit::isLegal() {
                    : cannotBeMaxInLoop(Start, L, *SE, InductionIsSigned))) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE
                ": induction start at a type extreme, no room for a boundary\n");
-    return false;
+    return std::nullopt;
   }
 
-  return true;
+  return LoopSplitAnalysis{InductionEnd, InductionIsSigned, Descending};
+}
+
+std::optional<LoopSplit>
+LoopSplit::get(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT) {
+  std::optional<LoopSplitAnalysis> Analysis = analyzeLegality(L, LI, SE, DT);
+  if (!Analysis)
+    return std::nullopt;
+  return LoopSplit(L, LI, SE, DT, Analysis->InductionEnd,
+                   Analysis->InductionIsSigned, Analysis->Descending);
 }
 
 //===----------------------------------------------------------------------===//
 // Transform
 //===----------------------------------------------------------------------===//
 
-// Latch "keep iterating" predicate, comparing the induction PHI against the
-// partition end: `i < sel` ascending, `i > sel` descending.
-static ICmpInst::Predicate continuePredicate(bool Signed, bool Descending) {
-  ICmpInst::Predicate P = Descending ? ICmpInst::ICMP_UGT : ICmpInst::ICMP_ULT;
-  return Signed ? ICmpInst::getSignedPredicate(P) : P;
-}
-
-// Guard "enter this partition" predicate: the latch test made non-strict, so
-// `start <= sel` ascending and `start >= sel` descending.
-static ICmpInst::Predicate guardPredicate(bool Signed, bool Descending) {
-  return ICmpInst::getNonStrictPredicate(continuePredicate(Signed, Descending));
-}
-
 static void buildEntryGuard(BasicBlock *&Preheader, BasicBlock *&EntryGuard,
                             DominatorTree *DT, LoopInfo *LI);
 
 // Drive the whole transform: set up scratch state and run each phase in order.
 bool LoopSplit::split() {
   PHINode *Induction = L->getInductionVariable(*SE);
-  assert(Induction && "split() requires a successful isLegal()");
+  assert(Induction && "split() requires prior legality analysis");
   if (getNumPartitions() < 2)
     return false;
 
@@ -371,8 +386,8 @@ void LoopSplit::splitFinalExit(SplitState &S) {
   BasicBlock *OrigExit = Partitions[0].Exit;
 
   // Splitting at begin() moves everything into FinalExit; the exit block has no
-  // PHIs because isLegal() rejects escaping values. SplitBlock also re-parents
-  // the dominator-tree children of the exit onto FinalExit.
+  // PHIs because legality analysis rejects escaping values. SplitBlock also
+  // re-parents the dominator-tree children of the exit onto FinalExit.
   S.FinalExit = SplitBlock(OrigExit, OrigExit->begin(), DT, LI,
                            /*MSSAU=*/nullptr, "ls.final.exit");
 }
@@ -426,7 +441,7 @@ void LoopSplit::expandPartitionBounds(SplitState &S, SCEVExpander &Expander) {
 // (partition 0 reuses the original loop).
 void LoopSplit::clonePartitions(SplitState &S) {
   Function &F = *L->getHeader()->getParent();
-  LLVMContext &Ctx = F.getContext();
+  LLVMContext &Ctx = SE->getContext();
 
   const unsigned N = getNumPartitions();
   // Partition 0 reuses the original loop; clone the rest off its preheader.
@@ -503,7 +518,7 @@ void LoopSplit::chainPartitions(SplitState &S) {
   // Emit each guard, clamp each latch, and chain partitions; a skipped
   // partition falls through to the next guard.
   const unsigned N = getNumPartitions();
-  IRBuilder<> B(L->getHeader()->getContext());
+  IRBuilder<> B(SE->getContext());
 
   for (unsigned I = 0; I < N; ++I) {
     PartitionInfo &P = Partitions[I];
diff --git a/llvm/lib/Transforms/Utils/LoopSplitPass.cpp b/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
index 3624c2526cdc2..2f85c2a75b546 100644
--- a/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
+++ b/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
@@ -48,22 +48,23 @@ static cl::opt<unsigned> SplitDepth(
 // the transform. Returns true if the loop was split.
 static bool splitLoop(Loop *L, ScalarEvolution &SE, DominatorTree &DT,
                       LoopInfo &LI) {
-  LoopSplit LS(L, &LI, &SE, &DT);
-  if (!LS.isLegal()) {
+  std::optional<LoopSplit> LS = LoopSplit::get(L, &LI, &SE, &DT);
+  if (!LS) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop is not legal for splitting\n");
     return false;
   }
 
-  // isLegal() has already established this shape.
-  const SCEV *IndVarSCEV = SE.getSCEV(LS.getInductionVariable());
+  // Legality analysis has already established this shape.
+  const SCEV *IndVarSCEV = SE.getSCEV(LS->getInductionVariable());
   const SCEV *Start;
   const APInt *StepC;
   [[maybe_unused]] bool Matched = match(
       IndVarSCEV, m_scev_AffineAddRec(m_SCEV(Start), m_scev_APInt(StepC)));
-  assert(Matched && "isLegal() guarantees a unit-step affine induction");
+  assert(Matched && (StepC->isOne() || StepC->isAllOnes()) &&
+         "expected unit-step affine induction");
 
   const SCEV *BTC = SE.getBackedgeTakenCount(L);
-  const SCEV *End = LS.getInductionEnd();
+  const SCEV *End = LS->getInductionEnd();
   Type *Ty = Start->getType();
   unsigned BitWidth = Ty->getIntegerBitWidth();
   // The backedge-taken count is a separate expression and need not share the
@@ -90,23 +91,23 @@ static bool splitLoop(Loop *L, ScalarEvolution &SE, DominatorTree &DT,
   for (unsigned Offset : Offsets) {
     // Clamp into [1, BTC] so each boundary stays in the space; Start +/- BTC is
     // the last iteration. The umax reaches one past it when BTC is zero, which
-    // isLegal() proved representable.
+    // Legality analysis proved representable.
     const SCEV *Off = SE.getConstant(Ty, Offset);
     Off = SE.getUMaxExpr(One, SE.getUMinExpr(Off, Count));
     const SCEV *Point =
         Descending ? SE.getMinusSCEV(Start, Off) : SE.getAddExpr(Start, Off);
     const SCEV *PrevEnd =
         Descending ? SE.getAddExpr(Point, One) : SE.getMinusSCEV(Point, One);
-    LS.addPartition(PrevStart, PrevEnd);
+    LS->addPartition(PrevStart, PrevEnd);
     PrevStart = Point;
   }
   // The final partition runs to the iteration-space end.
-  LS.addPartition(PrevStart, End);
+  LS->addPartition(PrevStart, End);
 
-  if (LS.getNumPartitions() < 2)
+  if (LS->getNumPartitions() < 2)
     return false;
 
-  return LS.split();
+  return LS->split();
 }
 
 // Split the selected loops in \p F at the command-line offsets.
diff --git a/llvm/test/Transforms/LoopSplit/unsupported-forms.ll b/llvm/test/Transforms/LoopSplit/unsupported-forms.ll
index d94b456c59f50..7553c9e72b88a 100644
--- a/llvm/test/Transforms/LoopSplit/unsupported-forms.ll
+++ b/llvm/test/Transforms/LoopSplit/unsupported-forms.ll
@@ -288,7 +288,7 @@ exit:
 }
 
 ; rewriteLatch expects a conditional latch terminator. An unconditional backedge
-; has no latch compare, so isLegal() rejects the loop before splitting.
+; has no latch compare, so legality analysis rejects the loop before splitting.
 
 define void @uncond_latch() {
 ; CHECK-LABEL: define void @uncond_latch() {
diff --git a/llvm/test/Transforms/LoopSplit/wrapping-iteration-space.ll b/llvm/test/Transforms/LoopSplit/wrapping-iteration-space.ll
index e4295649f6659..888d611a86d98 100644
--- a/llvm/test/Transforms/LoopSplit/wrapping-iteration-space.ll
+++ b/llvm/test/Transforms/LoopSplit/wrapping-iteration-space.ll
@@ -137,3 +137,54 @@ exit:
 ret:
   ret void
 }
+
+; An eq/ne latch with both nsw and nuw needs a bounded SCEV range to pick
+; signedness. A small fixed trip count narrows the signed range, so splitting
+; still succeeds (contrast @ambiguous_signedness in unsupported-forms.ll).
+
+define void @nuw_nsw_bounded_still_splits(ptr %a) {
+; CHECK-LABEL: define void @nuw_nsw_bounded_still_splits(
+; CHECK-SAME: ptr [[A:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp slt i64 [[IV]], 2
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    br i1 true, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i64 [ 3, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; CHECK-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add nuw nsw i64 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[IV_LS1]], 10
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nuw nsw i64 %iv, 1
+  %c = icmp ne i64 %iv.next, 11
+  br i1 %c, label %loop, label %exit
+
+exit:
+  ret void
+}

>From 5745ac418af1bc3ac82dc91cc39d7a62cde81949 Mon Sep 17 00:00:00 2001
From: Ashutosh Nema <ashu1212 at gmail.com>
Date: Fri, 28 Aug 2026 15:49:40 +0530
Subject: [PATCH 5/9] [Transforms][Utils] LoopSplit review follow-ups

---
 llvm/lib/Transforms/Utils/LoopSplit.cpp     | 6 ++----
 llvm/lib/Transforms/Utils/LoopSplitPass.cpp | 9 +++++----
 2 files changed, 7 insertions(+), 8 deletions(-)

diff --git a/llvm/lib/Transforms/Utils/LoopSplit.cpp b/llvm/lib/Transforms/Utils/LoopSplit.cpp
index ff35753403259..6548470325d2e 100644
--- a/llvm/lib/Transforms/Utils/LoopSplit.cpp
+++ b/llvm/lib/Transforms/Utils/LoopSplit.cpp
@@ -187,13 +187,11 @@ static std::optional<bool> computeSignedness(ScalarEvolution &SE, Loop *L,
   return std::nullopt;
 }
 
-// Prove \p Pred between \p LHS and \p RHS on entry to \p L, retrying with the
-// loop guards folded in so a bound fixed by a dominating condition is seen.
+// Prove \p Pred between \p LHS and \p RHS on entry to \p L, with loop guards
+// folded in so a bound fixed by a dominating condition is seen.
 static bool isEntryGuardedByCond(ScalarEvolution &SE, Loop *L,
                                  ICmpInst::Predicate Pred, const SCEV *LHS,
                                  const SCEV *RHS) {
-  if (SE.isLoopEntryGuardedByCond(L, Pred, LHS, RHS))
-    return true;
   return SE.isLoopEntryGuardedByCond(L, Pred, SE.applyLoopGuards(LHS, L),
                                      SE.applyLoopGuards(RHS, L));
 }
diff --git a/llvm/lib/Transforms/Utils/LoopSplitPass.cpp b/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
index 2f85c2a75b546..1131bf1dcedb5 100644
--- a/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
+++ b/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
@@ -58,10 +58,11 @@ static bool splitLoop(Loop *L, ScalarEvolution &SE, DominatorTree &DT,
   const SCEV *IndVarSCEV = SE.getSCEV(LS->getInductionVariable());
   const SCEV *Start;
   const APInt *StepC;
-  [[maybe_unused]] bool Matched = match(
-      IndVarSCEV, m_scev_AffineAddRec(m_SCEV(Start), m_scev_APInt(StepC)));
-  assert(Matched && (StepC->isOne() || StepC->isAllOnes()) &&
-         "expected unit-step affine induction");
+  [[maybe_unused]] bool Matched =
+      match(IndVarSCEV,
+            m_scev_AffineAddRec(m_SCEV(Start), m_scev_APInt(StepC))) &&
+      (StepC->isOne() || StepC->isAllOnes());
+  assert(Matched && "expected unit-step affine induction");
 
   const SCEV *BTC = SE.getBackedgeTakenCount(L);
   const SCEV *End = LS->getInductionEnd();

>From 8da7ba6a5d977f77e719f7b2b86c4a5af87948fc Mon Sep 17 00:00:00 2001
From: Ashutosh Nema <ashu1212 at gmail.com>
Date: Mon, 31 Aug 2026 14:32:08 +0530
Subject: [PATCH 6/9] [Transforms][Utils] LoopSplit review follow-ups
 (legality, metadata)

---
 .../include/llvm/Transforms/Utils/LoopSplit.h | 25 ++++--
 llvm/lib/Transforms/Utils/LoopSplit.cpp       | 83 +++++++++++++------
 .../Transforms/LoopSplit/loop-metadata.ll     | 63 ++++++++++++++
 .../Transforms/LoopSplit/unsupported-forms.ll | 38 ++++++++-
 4 files changed, 177 insertions(+), 32 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopSplit/loop-metadata.ll

diff --git a/llvm/include/llvm/Transforms/Utils/LoopSplit.h b/llvm/include/llvm/Transforms/Utils/LoopSplit.h
index a667676b4f48f..cd2771189a97a 100644
--- a/llvm/include/llvm/Transforms/Utils/LoopSplit.h
+++ b/llvm/include/llvm/Transforms/Utils/LoopSplit.h
@@ -59,7 +59,9 @@ class LoopSplit {
   /// Append an inclusive partition range [Start, End] in iteration order.
   /// Partitions must tile the whole space: first Start = induction start, each
   /// later Start = previous End +/- step, last End = induction end (desc: S >=
-  /// E).
+  /// E). For partition 0, \p Start must equal the induction start: that
+  /// partition reuses the original loop and does not reseed the induction PHI
+  /// (only the guard uses \p Start).
   ///
   /// Both bounds must have the induction type and be loop-invariant. They must
   /// also stay within the iteration space, extended by the one step past its
@@ -77,9 +79,11 @@ class LoopSplit {
 
 private:
   LoopSplit(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT,
-            const SCEV *InductionEnd, bool InductionIsSigned, bool Descending)
-      : L(L), LI(LI), SE(SE), DT(DT), InductionEnd(InductionEnd),
-        InductionIsSigned(InductionIsSigned), Descending(Descending) {}
+            const SCEV *InductionStart, const SCEV *InductionEnd,
+            bool InductionIsSigned, bool Descending)
+      : L(L), LI(LI), SE(SE), DT(DT), InductionStart(InductionStart),
+        InductionEnd(InductionEnd), InductionIsSigned(InductionIsSigned),
+        Descending(Descending) {}
 
   /// Everything known about one partition: the caller-supplied range plus the
   /// state split() derives. Indexed by partition number in \c Partitions.
@@ -112,13 +116,20 @@ class LoopSplit {
   DominatorTree *DT;
 
   // Induction analysis, populated during legality analysis.
-  const SCEV *InductionEnd = nullptr; // value on the last iteration.
-  bool InductionIsSigned = false;     // iteration ordering signedness.
-  bool Descending = false;            // step is -1 (the loop counts down).
+  const SCEV *InductionStart = nullptr; // value on the first iteration.
+  const SCEV *InductionEnd = nullptr;   // value on the last iteration.
+  bool InductionIsSigned = false;       // iteration ordering signedness.
+  bool Descending = false;              // step is -1 (the loop counts down).
 
   /// One record per partition, in add order.
   SmallVector<PartitionInfo, 4> Partitions;
 
+  /// Clamp \p EndExpr to the induction end for latch/guard materialization.
+  const SCEV *getClampedEndSCEV(const SCEV *EndExpr) const;
+  /// Return true if every partition bound is safe to expand at \p InsertPt.
+  bool arePartitionBoundsSafeToExpand(SCEVExpander &Expander,
+                                      Instruction *InsertPt) const;
+
   // split() phase helpers, run in order; each is documented at its definition.
   /// Split the final exit off the loop exit block.
   void splitFinalExit(SplitState &S);
diff --git a/llvm/lib/Transforms/Utils/LoopSplit.cpp b/llvm/lib/Transforms/Utils/LoopSplit.cpp
index 6548470325d2e..55e88a0a17c97 100644
--- a/llvm/lib/Transforms/Utils/LoopSplit.cpp
+++ b/llvm/lib/Transforms/Utils/LoopSplit.cpp
@@ -102,6 +102,7 @@ struct LoopSplit::SplitState {
   BasicBlock *FinalExit = nullptr; // where the partition chain converges.
   Loop *OuterLoop = nullptr;       // parent of the new blocks, if any.
   PHINode *Induction = nullptr;    // the loop's induction variable.
+  MDNode *OrigLoopID = nullptr;    // !llvm.loop on the loop before splitting.
 };
 
 // Record a new partition with the given inclusive iteration range.
@@ -113,6 +114,11 @@ void LoopSplit::addPartition(const SCEV *Start, const SCEV *End) {
   assert(Start->getType() == InductionEnd->getType() &&
          End->getType() == InductionEnd->getType() &&
          "partition bounds must have the induction type");
+  if (Partitions.empty()) {
+    assert((Start == InductionStart ||
+            SE->isKnownPredicate(ICmpInst::ICMP_EQ, Start, InductionStart)) &&
+           "first partition Start must match the induction start");
+  }
   Partitions.emplace_back(Start, End);
 }
 
@@ -211,6 +217,7 @@ static ICmpInst::Predicate guardPredicate(bool Signed, bool Descending) {
 }
 
 struct LoopSplitAnalysis {
+  const SCEV *InductionStart;
   const SCEV *InductionEnd;
   bool InductionIsSigned;
   bool Descending;
@@ -243,17 +250,14 @@ analyzeLegality(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop has exit values\n");
     return std::nullopt;
   }
-  for (BasicBlock *BB : L->blocks())
-    for (Instruction &I : *BB)
-      for (User *U : I.users())
-        if (auto *UI = dyn_cast<Instruction>(U); UI && !L->contains(UI)) {
-          LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop has exit values\n");
-          return std::nullopt;
-        }
-
-  // Splitting a loop clones it, so cloning must be safe.
-  if (!L->isSafeToClone()) {
-    LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop not safe to clone\n");
+  if (!findDefsUsedOutsideOfLoop(L).empty()) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop has exit values\n");
+    return std::nullopt;
+  }
+
+  // Guard-gated clones require isSafeToCloneConditionally().
+  if (!L->isSafeToCloneConditionally(*DT)) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE ": loop not safe to clone conditionally\n");
     return std::nullopt;
   }
 
@@ -322,7 +326,7 @@ analyzeLegality(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT) {
     return std::nullopt;
   }
 
-  return LoopSplitAnalysis{InductionEnd, InductionIsSigned, Descending};
+  return LoopSplitAnalysis{Start, InductionEnd, InductionIsSigned, Descending};
 }
 
 std::optional<LoopSplit>
@@ -330,8 +334,9 @@ LoopSplit::get(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT) {
   std::optional<LoopSplitAnalysis> Analysis = analyzeLegality(L, LI, SE, DT);
   if (!Analysis)
     return std::nullopt;
-  return LoopSplit(L, LI, SE, DT, Analysis->InductionEnd,
-                   Analysis->InductionIsSigned, Analysis->Descending);
+  return LoopSplit(L, LI, SE, DT, Analysis->InductionStart,
+                   Analysis->InductionEnd, Analysis->InductionIsSigned,
+                   Analysis->Descending);
 }
 
 //===----------------------------------------------------------------------===//
@@ -341,6 +346,25 @@ LoopSplit::get(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT) {
 static void buildEntryGuard(BasicBlock *&Preheader, BasicBlock *&EntryGuard,
                             DominatorTree *DT, LoopInfo *LI);
 
+const SCEV *LoopSplit::getClampedEndSCEV(const SCEV *EndExpr) const {
+  if (Descending)
+    return InductionIsSigned ? SE->getSMaxExpr(EndExpr, InductionEnd)
+                             : SE->getUMaxExpr(EndExpr, InductionEnd);
+  return InductionIsSigned ? SE->getSMinExpr(EndExpr, InductionEnd)
+                           : SE->getUMinExpr(EndExpr, InductionEnd);
+}
+
+bool LoopSplit::arePartitionBoundsSafeToExpand(SCEVExpander &Expander,
+                                               Instruction *InsertPt) const {
+  for (const PartitionInfo &P : Partitions) {
+    if (!Expander.isSafeToExpandAt(P.StartExpr, InsertPt))
+      return false;
+    if (!Expander.isSafeToExpandAt(getClampedEndSCEV(P.EndExpr), InsertPt))
+      return false;
+  }
+  return true;
+}
+
 // Drive the whole transform: set up scratch state and run each phase in order.
 bool LoopSplit::split() {
   PHINode *Induction = L->getInductionVariable(*SE);
@@ -348,6 +372,15 @@ bool LoopSplit::split() {
   if (getNumPartitions() < 2)
     return false;
 
+  // Check expansion safety at the preheader terminator before any CFG change.
+  Instruction *ExpandAt = L->getLoopPreheader()->getTerminator();
+  SCEVExpander Expander(*SE, DEBUG_TYPE);
+  if (!arePartitionBoundsSafeToExpand(Expander, ExpandAt)) {
+    LLVM_DEBUG(dbgs() << DEBUG_TYPE
+                      << ": partition bounds not safe to expand\n");
+    return false;
+  }
+
   SplitState S;
   // Partition 0 reuses the original loop; record its preheader/exit/guard up
   // front.
@@ -358,6 +391,7 @@ bool LoopSplit::split() {
   P0.IndPHI = Induction;
   S.OuterLoop = LI->getLoopFor(P0.Exit);
   S.Induction = Induction;
+  S.OrigLoopID = L->getLoopID();
 
   splitFinalExit(S);
   buildEntryGuard(P0.Preheader, P0.GuardBlock, DT, LI);
@@ -365,7 +399,6 @@ bool LoopSplit::split() {
   // Keep the expander and its cleaner alive for the whole transform: the bounds
   // it materializes are consumed by later phases. markResultUsed() below keeps
   // them; without it the cleaner reclaims them.
-  SCEVExpander Expander(*SE, DEBUG_TYPE);
   SCEVExpanderCleaner ExpanderCleaner(Expander);
   expandPartitionBounds(S, Expander);
   clonePartitions(S);
@@ -422,15 +455,7 @@ void LoopSplit::expandPartitionBounds(SplitState &S, SCEVExpander &Expander) {
 
     // Clamp the end to the induction end (min ascending, max descending) so a
     // short trip count keeps the last iteration in the right partition.
-    const SCEV *ClampedEndSCEV;
-    if (Descending)
-      ClampedEndSCEV = InductionIsSigned
-                           ? SE->getSMaxExpr(P.EndExpr, InductionEnd)
-                           : SE->getUMaxExpr(P.EndExpr, InductionEnd);
-    else
-      ClampedEndSCEV = InductionIsSigned
-                           ? SE->getSMinExpr(P.EndExpr, InductionEnd)
-                           : SE->getUMinExpr(P.EndExpr, InductionEnd);
+    const SCEV *ClampedEndSCEV = getClampedEndSCEV(P.EndExpr);
     P.SelEnd = Expander.expandCodeFor(ClampedEndSCEV, IndTy, EntryGuardTerm);
   }
 }
@@ -507,6 +532,15 @@ static void rewriteLatch(Loop *PL, PHINode *IndPHI, Value *SelEnd,
     Cmp->eraseFromParent();
 }
 
+// Attach a distinct !llvm.loop that inherits \p OrigLoopID's attributes.
+static void assignPartitionLoopID(Loop *PL, MDNode *OrigLoopID) {
+  if (!OrigLoopID)
+    return;
+  MDNode *NewLoopID = makePostTransformationMetadata(
+      PL->getHeader()->getContext(), OrigLoopID, {}, {});
+  PL->setLoopID(NewLoopID);
+}
+
 // Emit each partition's guard branch, clamp its latch, wire the partitions into
 // a chain, and update the dominator tree.
 void LoopSplit::chainPartitions(SplitState &S) {
@@ -539,6 +573,7 @@ void LoopSplit::chainPartitions(SplitState &S) {
 
     rewriteLatch(P.SubLoop, P.IndPHI, P.SelEnd, P.Exit, InductionIsSigned,
                  Descending);
+    assignPartitionLoopID(P.SubLoop, S.OrigLoopID);
     P.Exit->getTerminator()->setSuccessor(0, MergeAfter);
   }
 
diff --git a/llvm/test/Transforms/LoopSplit/loop-metadata.ll b/llvm/test/Transforms/LoopSplit/loop-metadata.ll
new file mode 100644
index 0000000000000..e3b468318af86
--- /dev/null
+++ b/llvm/test/Transforms/LoopSplit/loop-metadata.ll
@@ -0,0 +1,63 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6 --function split_preserves_loop_metadata -p
+; RUN: opt -passes=loop-split -loop-split-points=50 -S < %s | FileCheck %s
+
+; Each partition latch must carry !llvm.loop metadata derived from the original loop.
+
+define void @split_preserves_loop_metadata(ptr %a, i64 %n) {
+; CHECK-LABEL: define void @split_preserves_loop_metadata(ptr %a, i64 %n) {
+; CHECK-NEXT:  ls.guard0:
+; CHECK-NEXT:    %smax = call i64 @llvm.smax.i64(i64 %n, i64 1)
+; CHECK-NEXT:    %0 = add nsw i64 %smax, -1
+; CHECK-NEXT:    %umin = call i64 @llvm.umin.i64(i64 %0, i64 50)
+; CHECK-NEXT:    %1 = call i64 @llvm.usub.sat.i64(i64 %umin, i64 1)
+; CHECK-NEXT:    %smin = call i64 @llvm.smin.i64(i64 %0, i64 %1)
+; CHECK-NEXT:    %umax = call i64 @llvm.umax.i64(i64 %umin, i64 1)
+; CHECK-NEXT:    %itr.chk = icmp sle i64 0, %smin
+; CHECK-NEXT:    br i1 %itr.chk, label %entry, label %ls.guard1
+; CHECK:       entry:
+; CHECK-NEXT:    br label %loop
+; CHECK:       loop:
+; CHECK-NEXT:    %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+; CHECK-NEXT:    %p = getelementptr inbounds i64, ptr %a, i64 %iv
+; CHECK-NEXT:    store i64 %iv, ptr %p, align 4
+; CHECK-NEXT:    %iv.next = add nsw i64 %iv, 1
+; CHECK-NEXT:    %itr.chk1 = icmp slt i64 %iv, %smin
+; CHECK-NEXT:    br i1 %itr.chk1, label %loop, label %exit, !llvm.loop ![[LOOP0:[0-9]+]]
+; CHECK:       exit:
+; CHECK-NEXT:    br label %ls.guard1
+; CHECK:       ls.guard1:
+; CHECK-NEXT:    %itr.chk2 = icmp sle i64 %umax, %0
+; CHECK-NEXT:    br i1 %itr.chk2, label %entry.ls1, label %ls.final.exit
+; CHECK:       entry.ls1:
+; CHECK-NEXT:    br label %loop.ls1
+; CHECK:       loop.ls1:
+; CHECK-NEXT:    %iv.ls1 = phi i64 [ %umax, %entry.ls1 ], [ %iv.next.ls1, %loop.ls1 ]
+; CHECK-NEXT:    %p.ls1 = getelementptr inbounds i64, ptr %a, i64 %iv.ls1
+; CHECK-NEXT:    store i64 %iv.ls1, ptr %p.ls1, align 4
+; CHECK-NEXT:    %iv.next.ls1 = add nsw i64 %iv.ls1, 1
+; CHECK-NEXT:    %itr.chk3 = icmp slt i64 %iv.ls1, %0
+; CHECK-NEXT:    br i1 %itr.chk3, label %loop.ls1, label %ls.exit1, !llvm.loop ![[LOOP1:[0-9]+]]
+; CHECK:       ls.exit1:
+; CHECK-NEXT:    br label %ls.final.exit
+; CHECK:       ls.final.exit:
+; CHECK-NEXT:    ret void
+; CHECK:       ![[LOOP0]] = distinct !{![[LOOP0]], ![[ATTR:[0-9]+]]}
+; CHECK-NEXT:  ![[ATTR]] = !{!"llvm.loop.unroll.disable"}
+; CHECK-NEXT:  ![[LOOP1]] = distinct !{![[LOOP1]], ![[ATTR]]}
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
+  %p = getelementptr inbounds i64, ptr %a, i64 %iv
+  store i64 %iv, ptr %p
+  %iv.next = add nsw i64 %iv, 1
+  %c = icmp slt i64 %iv.next, %n
+  br i1 %c, label %loop, label %exit, !llvm.loop !0
+
+exit:
+  ret void
+}
+
+!0 = !{!0, !{!"llvm.loop.unroll.disable"}}
diff --git a/llvm/test/Transforms/LoopSplit/unsupported-forms.ll b/llvm/test/Transforms/LoopSplit/unsupported-forms.ll
index 7553c9e72b88a..bb8e49d23ef40 100644
--- a/llvm/test/Transforms/LoopSplit/unsupported-forms.ll
+++ b/llvm/test/Transforms/LoopSplit/unsupported-forms.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
 ; RUN: opt -passes=loop-split -loop-split-points=3 -S < %s | FileCheck %s
 
 ; Every loop here fails a legality check, so the pass must leave it alone. New
@@ -605,6 +605,40 @@ exit:
   ret void
 }
 
+declare i32 @conv(i32) #1
+
+; Guard-gated clones require isSafeToCloneConditionally().
+
+define void @cannot_clone_conditionally_convergent(i32 %n) {
+; CHECK-LABEL: define void @cannot_clone_conditionally_convergent(
+; CHECK-SAME: i32 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[C:%.*]] = call i32 @conv(i32 [[IV]])
+; CHECK-NEXT:    call void @use(i32 [[C]])
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], 1
+; CHECK-NEXT:    [[CMP:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[CMP]], label %[[LOOP]], label %[[EXIT:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %c = call i32 @conv(i32 %iv)
+  call void @use(i32 %c)
+  %iv.next = add nsw i32 %iv, 1
+  %cmp = icmp slt i32 %iv.next, %n
+  br i1 %cmp, label %loop, label %exit
+
+exit:
+  ret void
+}
+
 define void @cannot_clone_indirectbr(i32 %n, i1 %which) {
 ; CHECK-LABEL: define void @cannot_clone_indirectbr(
 ; CHECK-SAME: i32 [[N:%.*]], i1 [[WHICH:%.*]]) {
@@ -931,3 +965,5 @@ loop:
 exit:
   ret void
 }
+
+attributes #1 = { convergent }

>From 160c30aea0f3f59c6a85a81a8bfb914b131cb210 Mon Sep 17 00:00:00 2001
From: Ashutosh Nema <ashu1212 at gmail.com>
Date: Tue, 1 Sep 2026 11:30:51 +0530
Subject: [PATCH 7/9] [Transforms][Utils] LoopSplit: fix lit tests and header
 comments

---
 .../include/llvm/Transforms/Utils/LoopSplit.h | 14 ++--
 .../LoopSplit/constant-trip-count.ll          | 62 --------------
 .../Transforms/LoopSplit/loop-metadata.ll     | 81 ++++++++++---------
 3 files changed, 50 insertions(+), 107 deletions(-)

diff --git a/llvm/include/llvm/Transforms/Utils/LoopSplit.h b/llvm/include/llvm/Transforms/Utils/LoopSplit.h
index cd2771189a97a..6205a6b4fc7d9 100644
--- a/llvm/include/llvm/Transforms/Utils/LoopSplit.h
+++ b/llvm/include/llvm/Transforms/Utils/LoopSplit.h
@@ -115,11 +115,15 @@ class LoopSplit {
   ScalarEvolution *SE;
   DominatorTree *DT;
 
-  // Induction analysis, populated during legality analysis.
-  const SCEV *InductionStart = nullptr; // value on the first iteration.
-  const SCEV *InductionEnd = nullptr;   // value on the last iteration.
-  bool InductionIsSigned = false;       // iteration ordering signedness.
-  bool Descending = false;              // step is -1 (the loop counts down).
+  /// IV SCEV on the first iteration.
+  const SCEV *InductionStart = nullptr;
+  /// IV SCEV on the last iteration (from backedge-taken count).
+  const SCEV *InductionEnd = nullptr;
+  /// True if the original loop treats the IV as signed, false if unsigned.
+  /// Partition guards and end bounds follow the same signedness.
+  bool InductionIsSigned = false;
+  /// Induction step is -1 (loop counts down).
+  bool Descending = false;
 
   /// One record per partition, in add order.
   SmallVector<PartitionInfo, 4> Partitions;
diff --git a/llvm/test/Transforms/LoopSplit/constant-trip-count.ll b/llvm/test/Transforms/LoopSplit/constant-trip-count.ll
index 9cdabd0056539..f3c7c94b2f7ea 100644
--- a/llvm/test/Transforms/LoopSplit/constant-trip-count.ll
+++ b/llvm/test/Transforms/LoopSplit/constant-trip-count.ll
@@ -13,68 +13,6 @@
 ; computed from scratch; later passes delete the dead partition.
 
 define void @tc100(ptr %a) {
-; IN-LABEL: define void @tc100(
-; IN-SAME: ptr [[A:%.*]]) {
-; IN-NEXT:  [[LS_GUARD0:.*:]]
-; IN-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
-; IN:       [[ENTRY]]:
-; IN-NEXT:    br label %[[LOOP:.*]]
-; IN:       [[LOOP]]:
-; IN-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; IN-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
-; IN-NEXT:    store i64 [[IV]], ptr [[P]], align 4
-; IN-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
-; IN-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 [[IV_NEXT]], 49
-; IN-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
-; IN:       [[EXIT]]:
-; IN-NEXT:    br label %[[LS_GUARD1]]
-; IN:       [[LS_GUARD1]]:
-; IN-NEXT:    br i1 true, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
-; IN:       [[ENTRY_LS1]]:
-; IN-NEXT:    br label %[[LOOP_LS1:.*]]
-; IN:       [[LOOP_LS1]]:
-; IN-NEXT:    [[IV_LS1:%.*]] = phi i64 [ 50, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
-; IN-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
-; IN-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
-; IN-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], 1
-; IN-NEXT:    [[ITR_CHK1:%.*]] = icmp sle i64 [[IV_NEXT_LS1]], 99
-; IN-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
-; IN:       [[LS_EXIT1]]:
-; IN-NEXT:    br label %[[LS_FINAL_EXIT]]
-; IN:       [[LS_FINAL_EXIT]]:
-; IN-NEXT:    ret void
-;
-; OUT-LABEL: define void @tc100(
-; OUT-SAME: ptr [[A:%.*]]) {
-; OUT-NEXT:  [[LS_GUARD0:.*:]]
-; OUT-NEXT:    br i1 true, label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
-; OUT:       [[ENTRY]]:
-; OUT-NEXT:    br label %[[LOOP:.*]]
-; OUT:       [[LOOP]]:
-; OUT-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
-; OUT-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
-; OUT-NEXT:    store i64 [[IV]], ptr [[P]], align 4
-; OUT-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
-; OUT-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 [[IV_NEXT]], 99
-; OUT-NEXT:    br i1 [[ITR_CHK]], label %[[LOOP]], label %[[EXIT:.*]]
-; OUT:       [[EXIT]]:
-; OUT-NEXT:    br label %[[LS_GUARD1]]
-; OUT:       [[LS_GUARD1]]:
-; OUT-NEXT:    br i1 false, label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
-; OUT:       [[ENTRY_LS1]]:
-; OUT-NEXT:    br label %[[LOOP_LS1:.*]]
-; OUT:       [[LOOP_LS1]]:
-; OUT-NEXT:    [[IV_LS1:%.*]] = phi i64 [ 200, %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
-; OUT-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
-; OUT-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
-; OUT-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], 1
-; OUT-NEXT:    [[ITR_CHK1:%.*]] = icmp sle i64 [[IV_NEXT_LS1]], 99
-; OUT-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]]
-; OUT:       [[LS_EXIT1]]:
-; OUT-NEXT:    br label %[[LS_FINAL_EXIT]]
-; OUT:       [[LS_FINAL_EXIT]]:
-; OUT-NEXT:    ret void
-;
 ; INSIDE-LABEL: define void @tc100(
 ; INSIDE-SAME: ptr [[A:%.*]]) {
 ; INSIDE-NEXT:  [[LS_GUARD0:.*:]]
diff --git a/llvm/test/Transforms/LoopSplit/loop-metadata.ll b/llvm/test/Transforms/LoopSplit/loop-metadata.ll
index e3b468318af86..b138d34a636f8 100644
--- a/llvm/test/Transforms/LoopSplit/loop-metadata.ll
+++ b/llvm/test/Transforms/LoopSplit/loop-metadata.ll
@@ -1,49 +1,50 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6 --function split_preserves_loop_metadata -p
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
 ; RUN: opt -passes=loop-split -loop-split-points=50 -S < %s | FileCheck %s
 
 ; Each partition latch must carry !llvm.loop metadata derived from the original loop.
 
 define void @split_preserves_loop_metadata(ptr %a, i64 %n) {
-; CHECK-LABEL: define void @split_preserves_loop_metadata(ptr %a, i64 %n) {
-; CHECK-NEXT:  ls.guard0:
-; CHECK-NEXT:    %smax = call i64 @llvm.smax.i64(i64 %n, i64 1)
-; CHECK-NEXT:    %0 = add nsw i64 %smax, -1
-; CHECK-NEXT:    %umin = call i64 @llvm.umin.i64(i64 %0, i64 50)
-; CHECK-NEXT:    %1 = call i64 @llvm.usub.sat.i64(i64 %umin, i64 1)
-; CHECK-NEXT:    %smin = call i64 @llvm.smin.i64(i64 %0, i64 %1)
-; CHECK-NEXT:    %umax = call i64 @llvm.umax.i64(i64 %umin, i64 1)
-; CHECK-NEXT:    %itr.chk = icmp sle i64 0, %smin
-; CHECK-NEXT:    br i1 %itr.chk, label %entry, label %ls.guard1
-; CHECK:       entry:
-; CHECK-NEXT:    br label %loop
-; CHECK:       loop:
-; CHECK-NEXT:    %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop ]
-; CHECK-NEXT:    %p = getelementptr inbounds i64, ptr %a, i64 %iv
-; CHECK-NEXT:    store i64 %iv, ptr %p, align 4
-; CHECK-NEXT:    %iv.next = add nsw i64 %iv, 1
-; CHECK-NEXT:    %itr.chk1 = icmp slt i64 %iv, %smin
-; CHECK-NEXT:    br i1 %itr.chk1, label %loop, label %exit, !llvm.loop ![[LOOP0:[0-9]+]]
-; CHECK:       exit:
-; CHECK-NEXT:    br label %ls.guard1
-; CHECK:       ls.guard1:
-; CHECK-NEXT:    %itr.chk2 = icmp sle i64 %umax, %0
-; CHECK-NEXT:    br i1 %itr.chk2, label %entry.ls1, label %ls.final.exit
-; CHECK:       entry.ls1:
-; CHECK-NEXT:    br label %loop.ls1
-; CHECK:       loop.ls1:
-; CHECK-NEXT:    %iv.ls1 = phi i64 [ %umax, %entry.ls1 ], [ %iv.next.ls1, %loop.ls1 ]
-; CHECK-NEXT:    %p.ls1 = getelementptr inbounds i64, ptr %a, i64 %iv.ls1
-; CHECK-NEXT:    store i64 %iv.ls1, ptr %p.ls1, align 4
-; CHECK-NEXT:    %iv.next.ls1 = add nsw i64 %iv.ls1, 1
-; CHECK-NEXT:    %itr.chk3 = icmp slt i64 %iv.ls1, %0
-; CHECK-NEXT:    br i1 %itr.chk3, label %loop.ls1, label %ls.exit1, !llvm.loop ![[LOOP1:[0-9]+]]
-; CHECK:       ls.exit1:
-; CHECK-NEXT:    br label %ls.final.exit
-; CHECK:       ls.final.exit:
+; CHECK-LABEL: define void @split_preserves_loop_metadata(
+; CHECK-SAME: ptr [[A:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[LS_GUARD0:.*:]]
+; CHECK-NEXT:    [[SMAX:%.*]] = call i64 @llvm.smax.i64(i64 [[N]], i64 1)
+; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i64 [[SMAX]], -1
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 50)
+; CHECK-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[UMIN]], i64 1)
+; CHECK-NEXT:    [[SMIN:%.*]] = call i64 @llvm.smin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[UMIN]], i64 1)
+; CHECK-NEXT:    [[ITR_CHK:%.*]] = icmp sle i64 0, [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK]], label %[[ENTRY:.*]], label %[[LS_GUARD1:.*]]
+; CHECK:       [[ENTRY]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[P:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV]]
+; CHECK-NEXT:    store i64 [[IV]], ptr [[P]], align 4
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i64 [[IV]], 1
+; CHECK-NEXT:    [[ITR_CHK1:%.*]] = icmp slt i64 [[IV]], [[SMIN]]
+; CHECK-NEXT:    br i1 [[ITR_CHK1]], label %[[LOOP]], label %[[EXIT:.*]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    br label %[[LS_GUARD1]]
+; CHECK:       [[LS_GUARD1]]:
+; CHECK-NEXT:    [[ITR_CHK2:%.*]] = icmp sle i64 [[UMAX]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK2]], label %[[ENTRY_LS1:.*]], label %[[LS_FINAL_EXIT:.*]]
+; CHECK:       [[ENTRY_LS1]]:
+; CHECK-NEXT:    br label %[[LOOP_LS1:.*]]
+; CHECK:       [[LOOP_LS1]]:
+; CHECK-NEXT:    [[IV_LS1:%.*]] = phi i64 [ [[UMAX]], %[[ENTRY_LS1]] ], [ [[IV_NEXT_LS1:%.*]], %[[LOOP_LS1]] ]
+; CHECK-NEXT:    [[P_LS1:%.*]] = getelementptr inbounds i64, ptr [[A]], i64 [[IV_LS1]]
+; CHECK-NEXT:    store i64 [[IV_LS1]], ptr [[P_LS1]], align 4
+; CHECK-NEXT:    [[IV_NEXT_LS1]] = add nsw i64 [[IV_LS1]], 1
+; CHECK-NEXT:    [[ITR_CHK3:%.*]] = icmp slt i64 [[IV_LS1]], [[TMP0]]
+; CHECK-NEXT:    br i1 [[ITR_CHK3]], label %[[LOOP_LS1]], label %[[LS_EXIT1:.*]], !llvm.loop [[LOOP2:![0-9]+]]
+; CHECK:       [[LS_EXIT1]]:
+; CHECK-NEXT:    br label %[[LS_FINAL_EXIT]]
+; CHECK:       [[LS_FINAL_EXIT]]:
 ; CHECK-NEXT:    ret void
-; CHECK:       ![[LOOP0]] = distinct !{![[LOOP0]], ![[ATTR:[0-9]+]]}
-; CHECK-NEXT:  ![[ATTR]] = !{!"llvm.loop.unroll.disable"}
-; CHECK-NEXT:  ![[LOOP1]] = distinct !{![[LOOP1]], ![[ATTR]]}
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[ATTR:![0-9]+]]}
+; CHECK: [[ATTR]] = !{!"llvm.loop.unroll.disable"}
+; CHECK: [[LOOP2]] = distinct !{[[LOOP2]], [[ATTR]]}
 ;
 entry:
   br label %loop

>From a6c8f372f7d5f9481fb91805bcc0ffb05109e814 Mon Sep 17 00:00:00 2001
From: Ashutosh Nema <ashu1212 at gmail.com>
Date: Tue, 22 Sep 2026 10:45:44 +0530
Subject: [PATCH 8/9] LoopSplit - Addressed Review comments

---
 .../include/llvm/Transforms/Utils/LoopSplit.h |  5 --
 llvm/lib/Transforms/Utils/LoopSplit.cpp       | 74 ++++++-------------
 llvm/lib/Transforms/Utils/LoopSplitPass.cpp   |  2 +-
 3 files changed, 25 insertions(+), 56 deletions(-)

diff --git a/llvm/include/llvm/Transforms/Utils/LoopSplit.h b/llvm/include/llvm/Transforms/Utils/LoopSplit.h
index 6205a6b4fc7d9..4826e76866ac7 100644
--- a/llvm/include/llvm/Transforms/Utils/LoopSplit.h
+++ b/llvm/include/llvm/Transforms/Utils/LoopSplit.h
@@ -47,11 +47,6 @@ class LoopSplit {
   LLVM_ABI static std::optional<LoopSplit>
   get(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT);
 
-  /// Return the loop's induction variable. Valid only on a legal LoopSplit.
-  LLVM_ABI PHINode *getInductionVariable() const {
-    return L->getInductionVariable(*SE);
-  }
-
   /// The induction value on the last iteration, which the final partition must
   /// end at. Valid only on a legal LoopSplit.
   LLVM_ABI const SCEV *getInductionEnd() const { return InductionEnd; }
diff --git a/llvm/lib/Transforms/Utils/LoopSplit.cpp b/llvm/lib/Transforms/Utils/LoopSplit.cpp
index 55e88a0a17c97..6f326599ff30f 100644
--- a/llvm/lib/Transforms/Utils/LoopSplit.cpp
+++ b/llvm/lib/Transforms/Utils/LoopSplit.cpp
@@ -95,14 +95,20 @@ using namespace llvm::SCEVPatternMatch;
 
 /// Per-split() scratch shared by the phase helpers; lives for one split() call.
 /// Everything derived from the induction lives on LoopSplit itself, filled in
-/// by legality analysis; this holds only what the transform creates.
+/// by legality analysis; this holds only what the transform creates. Partition
+/// 0 reuses the original loop's blocks, which live in Partitions[0].
 struct LoopSplit::SplitState {
-  // Partition 0 reuses the original loop's preheader, exit, and entry guard;
-  // those blocks live in Partitions[0] rather than being duplicated here.
-  BasicBlock *FinalExit = nullptr; // where the partition chain converges.
-  Loop *OuterLoop = nullptr;       // parent of the new blocks, if any.
-  PHINode *Induction = nullptr;    // the loop's induction variable.
-  MDNode *OrigLoopID = nullptr;    // !llvm.loop on the loop before splitting.
+  // Where the partition chain converges.
+  BasicBlock *FinalExit = nullptr;
+
+  // Parent of the new blocks, if any.
+  Loop *OuterLoop = nullptr;
+
+  // The loop's induction variable.
+  PHINode *Induction = nullptr;
+
+  // !llvm.loop on the loop before splitting.
+  MDNode *OrigLoopID = nullptr;
 };
 
 // Record a new partition with the given inclusive iteration range.
@@ -123,14 +129,9 @@ void LoopSplit::addPartition(const SCEV *Start, const SCEV *End) {
 }
 
 // Return the induction's add-recurrence, or null unless the induction is an
-// integer with a unit step that the latch compares.
-static const SCEVAddRecExpr *analyzeInduction(Loop *L, ScalarEvolution *SE) {
-  ICmpInst *LatchCmp = L->getLatchCmpInst();
-
-  // SCEV's induction variable, restricted to a unit-step affine recurrence.
-  PHINode *Induction = L->getInductionVariable(*SE);
-  if (!Induction)
-    return nullptr;
+// integer with a unit step.
+static const SCEVAddRecExpr *analyzeInduction(PHINode *Induction,
+                                              ScalarEvolution *SE) {
   // Partition bounds are integer arithmetic on the induction type, so a loop
   // whose only induction is a pointer is out of scope.
   if (!Induction->getType()->isIntegerTy())
@@ -143,20 +144,7 @@ static const SCEVAddRecExpr *analyzeInduction(Loop *L, ScalarEvolution *SE) {
     return nullptr;
   if (!Step->isOne() && !Step->isAllOnes())
     return nullptr;
-  const auto *AR = cast<SCEVAddRecExpr>(IndSCEV);
-
-  // The induction's "next" value (i + 1), produced in the latch.
-  auto *StepInst = dyn_cast<Instruction>(
-      Induction->getIncomingValueForBlock(L->getLoopLatch()));
-  if (!StepInst)
-    return nullptr;
-
-  // One compare operand must be the induction, either the PHI or its step. The
-  // rebuilt latch always compares the PHI, so which operand it was is not used.
-  if (any_of(LatchCmp->operands(),
-             [&](Value *Op) { return Op == Induction || Op == StepInst; }))
-    return AR;
-  return nullptr;
+  return cast<SCEVAddRecExpr>(IndSCEV);
 }
 
 // Decide whether the iteration ordering is signed or unsigned; returns the
@@ -171,11 +159,6 @@ static std::optional<bool> computeSignedness(ScalarEvolution &SE, Loop *L,
   if (IndAR->hasNoSignedWrap() && IndAR->hasNoUnsignedWrap()) {
     const ConstantRange UR = SE.getUnsignedRange(IndAR);
     const ConstantRange SR = SE.getSignedRange(IndAR);
-    if (UR.isFullSet() && SR.isFullSet()) {
-      LLVM_DEBUG(dbgs() << DEBUG_TYPE
-                 ": ambiguous iteration ordering with both nsw and nuw\n");
-      return std::nullopt;
-    }
     if (!SR.isFullSet())
       return true;
     if (!UR.isFullSet())
@@ -198,8 +181,9 @@ static std::optional<bool> computeSignedness(ScalarEvolution &SE, Loop *L,
 static bool isEntryGuardedByCond(ScalarEvolution &SE, Loop *L,
                                  ICmpInst::Predicate Pred, const SCEV *LHS,
                                  const SCEV *RHS) {
-  return SE.isLoopEntryGuardedByCond(L, Pred, SE.applyLoopGuards(LHS, L),
-                                     SE.applyLoopGuards(RHS, L));
+  auto Guards = ScalarEvolution::LoopGuards::collect(L, SE);
+  return SE.isLoopEntryGuardedByCond(L, Pred, SE.applyLoopGuards(LHS, Guards),
+                                     SE.applyLoopGuards(RHS, Guards));
 }
 
 // Latch "keep iterating" predicate, comparing the induction PHI against the
@@ -268,15 +252,15 @@ analyzeLegality(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT) {
     return std::nullopt;
   }
 
-  const SCEVAddRecExpr *IndAR = analyzeInduction(L, SE);
+  PHINode *Induction = L->getInductionVariable(*SE);
+  const SCEVAddRecExpr *IndAR =
+      Induction ? analyzeInduction(Induction, SE) : nullptr;
   if (!IndAR) {
     LLVM_DEBUG(dbgs() << DEBUG_TYPE
                ": no unique unit-step integer induction\n");
     return std::nullopt;
   }
 
-  PHINode *Induction = L->getInductionVariable(*SE);
-
   // Loop-carried values are unsupported: a later partition would have to resume
   // the previous one's value, which needs SSA reconstruction. The induction is
   // the exception, seeded per partition from its own start bound.
@@ -293,14 +277,7 @@ analyzeLegality(Loop *L, LoopInfo *LI, ScalarEvolution *SE, DominatorTree *DT) {
   const bool Descending =
       cast<SCEVConstant>(IndAR->getStepRecurrence(*SE))->getAPInt().isAllOnes();
 
-  // Start and end must share the induction type; reject any width mismatch.
-  // evaluateAtIteration coerces to the start's type for an affine recurrence,
-  // so this is defensive rather than reachable.
   const SCEV *InductionEnd = IndAR->evaluateAtIteration(BTC, *SE);
-  if (InductionEnd->getType() != IndAR->getStart()->getType()) {
-    LLVM_DEBUG(dbgs() << DEBUG_TYPE ": induction end/start type mismatch\n");
-    return std::nullopt;
-  }
 
   // Partition bounds and entry guards assume the space runs monotonically from
   // start to end, so refuse one that wraps past the type extreme. A no-wrap
@@ -447,10 +424,7 @@ void LoopSplit::expandPartitionBounds(SplitState &S, SCEVExpander &Expander) {
 
   // Expand all partition bounds in the entry guard, which dominates the whole
   // chain (a skipped partition bypasses the original preheader).
-  const unsigned N = getNumPartitions();
-  for (unsigned I = 0; I < N; ++I) {
-    PartitionInfo &P = Partitions[I];
-
+  for (PartitionInfo &P : Partitions) {
     P.StartVal = Expander.expandCodeFor(P.StartExpr, IndTy, EntryGuardTerm);
 
     // Clamp the end to the induction end (min ascending, max descending) so a
diff --git a/llvm/lib/Transforms/Utils/LoopSplitPass.cpp b/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
index 1131bf1dcedb5..ab5a8a5fe4257 100644
--- a/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
+++ b/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
@@ -55,7 +55,7 @@ static bool splitLoop(Loop *L, ScalarEvolution &SE, DominatorTree &DT,
   }
 
   // Legality analysis has already established this shape.
-  const SCEV *IndVarSCEV = SE.getSCEV(LS->getInductionVariable());
+  const SCEV *IndVarSCEV = SE.getSCEV(L->getInductionVariable(SE));
   const SCEV *Start;
   const APInt *StepC;
   [[maybe_unused]] bool Matched =

>From a3010b588f89c6f34f510ef9ff39f65f7024a97b Mon Sep 17 00:00:00 2001
From: Ashutosh Nema <ashu1212 at gmail.com>
Date: Tue, 22 Sep 2026 15:49:39 +0530
Subject: [PATCH 9/9] LoopSplit : Build issue fix

---
 llvm/lib/Transforms/Utils/LoopSplitPass.cpp | 13 +++++++++----
 1 file changed, 9 insertions(+), 4 deletions(-)

diff --git a/llvm/lib/Transforms/Utils/LoopSplitPass.cpp b/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
index ab5a8a5fe4257..6838f1beba9ff 100644
--- a/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
+++ b/llvm/lib/Transforms/Utils/LoopSplitPass.cpp
@@ -95,10 +95,15 @@ static bool splitLoop(Loop *L, ScalarEvolution &SE, DominatorTree &DT,
     // Legality analysis proved representable.
     const SCEV *Off = SE.getConstant(Ty, Offset);
     Off = SE.getUMaxExpr(One, SE.getUMinExpr(Off, Count));
-    const SCEV *Point =
-        Descending ? SE.getMinusSCEV(Start, Off) : SE.getAddExpr(Start, Off);
-    const SCEV *PrevEnd =
-        Descending ? SE.getAddExpr(Point, One) : SE.getMinusSCEV(Point, One);
+    const SCEV *Point;
+    const SCEV *PrevEnd;
+    if (Descending) {
+      Point = SE.getMinusSCEV(Start, Off);
+      PrevEnd = SE.getAddExpr(Point, One);
+    } else {
+      Point = SE.getAddExpr(Start, Off);
+      PrevEnd = SE.getMinusSCEV(Point, One);
+    }
     LS->addPartition(PrevStart, PrevEnd);
     PrevStart = Point;
   }



More information about the llvm-commits mailing list