[llvm] [IndVars] Rewrite loop-exit comparisons from operands (PR #223951)

Pengcheng Wang via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 20 00:34:41 PDT 2026


https://github.com/wangpc-pp updated https://github.com/llvm/llvm-project/pull/223951

>From 31532d528e1c67110a63616b53edddab203b1737 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 12:58:40 +0800
Subject: [PATCH 1/6] [IndVars] Add tests for rewriting loop-exit comparisons

Add coverage for integer and pointer comparisons, multiple varying

operands, uncomputable exit values, and the RISC-V phase-ordering

issue.

Assisted-by: TRAE CLI (GPT-5)
---
 .../IndVarSimplify/rewrite-loop-exit-value.ll | 194 ++++++++++++++++++
 .../RISCV/rewrite-loop-exit-compare.ll        |  70 +++++++
 2 files changed, 264 insertions(+)
 create mode 100644 llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll

diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index fa47d06d859e9..09e8765610f69 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -302,3 +302,197 @@ for.body:
 for.end:
   ret i32 %VF.capped
 }
+
+; Rewrite a comparison by expanding its operands at the loop exit.
+define i1 @rewrite_computable_icmp(i32 %start, i32 %limit) {
+; CHECK-LABEL: @rewrite_computable_icmp(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[IV]], 42
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP_EXIT]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP_EXIT]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp = icmp ne i32 %iv, 42
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %iv.next = add nsw i32 %iv, -1
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %cmp
+}
+
+; Rewrite multiple comparisons when doing so makes all live-outs invariant.
+define i1 @rewrite_multiple_icmps(i32 %start, i32 %rhs.start, i32 %limit) {
+; CHECK-LABEL: @rewrite_multiple_icmps(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[TMP2:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = phi i32 [ [[RHS_START:%.*]], [[ENTRY]] ], [ [[RHS_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP0_EXIT]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[TMP2]], -1
+; CHECK-NEXT:    [[RHS_NEXT]] = add i32 [[TMP4]], 2
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    [[CMP1_EXIT:%.*]] = icmp sgt i32 [[TMP2]], [[TMP4]]
+; CHECK-NEXT:    [[RESULT:%.*]] = and i1 [[CMP0_EXIT]], [[CMP1_EXIT]]
+; CHECK-NEXT:    ret i1 [[RESULT]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+  %rhs = phi i32 [ %rhs.start, %entry ], [ %rhs.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp0 = icmp ne i32 %iv, 42
+  %cmp1 = icmp sgt i32 %iv, %rhs
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp0, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %iv.next = add nsw i32 %iv, -1
+  %rhs.next = add i32 %rhs, 2
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  %result = and i1 %cmp0, %cmp1
+  ret i1 %result
+}
+
+; Do not rewrite a comparison if SCEV cannot compute an operand's exit value.
+define i1 @do_not_rewrite_uncomputable_icmp(i1 %c, i32 %start) {
+; CHECK-LABEL: @do_not_rewrite_uncomputable_icmp(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i32 [[IV]], 0
+; CHECK-NEXT:    [[SELECTED:%.*]] = select i1 [[CMP]], i1 [[C:%.*]], i1 false
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT:    br i1 [[SELECTED]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %loop ]
+  %cmp = icmp ne i32 %iv, 0
+  %selected = select i1 %cmp, i1 %c, i1 false
+  %iv.next = add i32 %iv, 1
+  br i1 %selected, label %loop, label %exit
+
+exit:
+  ret i1 %cmp
+}
+
+; Do not rebuild a comparison unless doing so makes the loop deletable.
+define i1 @do_not_rewrite_icmp_in_live_loop(i32 %start, i32 %limit) {
+; CHECK-LABEL: @do_not_rewrite_icmp_in_live_loop(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i32 [[IV]], 42
+; CHECK-NEXT:    call void @use.i1(i1 [[CMP]])
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    br i1 [[INRANGE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp = icmp ne i32 %iv, 42
+  call void @use.i1(i1 %cmp)
+  %inrange = icmp ult i32 %index, %limit
+  br i1 %inrange, label %body, label %exit
+
+body:
+  %iv.next = add nsw i32 %iv, -1
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %cmp
+}
+
+; Rewrite a pointer comparison when its operand exit values are computable.
+define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
+; CHECK-LABEL: @rewrite_pointer_icmp(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[PTR:%.*]] = phi ptr [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[PTR_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne ptr [[PTR]], [[END:%.*]]
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %ptr = phi ptr [ %start, %entry ], [ %ptr.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp = icmp ne ptr %ptr, %end
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %ptr.next = getelementptr i8, ptr %ptr, i64 1
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %cmp
+}
+
+declare void @use.i1(i1)
diff --git a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
new file mode 100644
index 0000000000000..f91972263d862
--- /dev/null
+++ b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
@@ -0,0 +1,70 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -mtriple=riscv64 -mattr=+v -passes='default<O3>' -S %s | FileCheck %s
+
+; Rewriting the comparison from its operands' exit values allows the loop to be
+; deleted instead of vectorizing the induction and extracting its final result.
+define i1 @rewrite_loop_exit_compare(i32 %length, i64 %limit) vscale_range(2, 1024) {
+; CHECK-LABEL: define i1 @rewrite_loop_exit_compare(
+; CHECK-SAME: i32 [[LENGTH:%.*]], i64 [[LIMIT:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[LIMIT_FR:%.*]] = freeze i64 [[LIMIT]]
+; CHECK-NEXT:    [[NONZERO1:%.*]] = icmp ne i32 [[LENGTH]], 0
+; CHECK-NEXT:    [[INRANGE2:%.*]] = icmp ne i64 [[LIMIT_FR]], 0
+; CHECK-NEXT:    [[CONTINUE3:%.*]] = and i1 [[NONZERO1]], [[INRANGE2]]
+; CHECK-NEXT:    br i1 [[CONTINUE3]], label %[[BODY_PREHEADER:.*]], label %[[EXIT:.*]]
+; CHECK:       [[BODY_PREHEADER]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
+; CHECK-NEXT:    [[TMP0:%.*]] = add i32 [[LENGTH]], -1
+; CHECK-NEXT:    [[TMP2:%.*]] = zext i32 [[TMP0]] to i64
+; CHECK-NEXT:    [[UMIN:%.*]] = tail call i64 @llvm.umin.i64(i64 [[TMP1]], i64 [[TMP2]])
+; CHECK-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[UMIN]], 1
+; CHECK-NEXT:    [[TMP5:%.*]] = tail call <vscale x 16 x i32> @llvm.stepvector.nxv16i32()
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[LENGTH]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP6:%.*]] = sub nsw <vscale x 16 x i32> [[BROADCAST_SPLAT]], [[TMP5]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 16 x i32> [ [[TMP6]], %[[BODY_PREHEADER]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[AVL:%.*]] = phi i64 [ [[TMP4]], %[[BODY_PREHEADER]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP7:%.*]] = tail call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 16, i1 true)
+; CHECK-NEXT:    [[TMP8:%.*]] = zext i32 [[TMP7]] to i64
+; CHECK-NEXT:    [[TMP9:%.*]] = sub nsw i32 0, [[TMP7]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT9:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[TMP9]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT10:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT9]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-NEXT:    [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP8]]
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nsw <vscale x 16 x i32> [[VEC_IND]], [[BROADCAST_SPLAT10]]
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[EXIT_LOOPEXIT:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[EXIT_LOOPEXIT]]:
+; CHECK-NEXT:    [[TMP11:%.*]] = add nsw i64 [[TMP8]], -1
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <vscale x 16 x i32> [[VEC_IND]], i64 [[TMP11]]
+; CHECK-NEXT:    [[TMP13:%.*]] = icmp ne i32 [[TMP12]], 1
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[TMP13]], %[[EXIT_LOOPEXIT]] ]
+; CHECK-NEXT:    ret i1 [[NONZERO_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %length.addr = phi i32 [ %length, %entry ], [ %dec, %body ]
+  %index = phi i64 [ 0, %entry ], [ %index.next, %body ]
+  %nonzero = icmp ne i32 %length.addr, 0
+  %inrange = icmp ult i64 %index, %limit
+  %continue = select i1 %nonzero, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %dec = add nsw i32 %length.addr, -1
+  %index.next = add nuw i64 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %nonzero
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+;.

>From 276cb73e2b17e56fb1d11db1d16f4eaef7895992 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 13:00:15 +0800
Subject: [PATCH 2/6] [IndVars] Rewrite loop-exit comparisons from operands

When SCEV cannot model a comparison, clone it in the loop preheader

using the exit values of its varying operands. This lets exit-value

rewriting remove otherwise dead loops before vectorization.

Assisted-by: TRAE CLI (GPT-5)
---
 llvm/lib/Transforms/Utils/LoopUtils.cpp       | 87 +++++++++++++++----
 .../IndVarSimplify/rewrite-loop-exit-value.ll | 49 +++++------
 .../RISCV/rewrite-loop-exit-compare.ll        | 33 +------
 3 files changed, 97 insertions(+), 72 deletions(-)

diff --git a/llvm/lib/Transforms/Utils/LoopUtils.cpp b/llvm/lib/Transforms/Utils/LoopUtils.cpp
index a2e544801b9c4..5c1a6fbaebbdd 100644
--- a/llvm/lib/Transforms/Utils/LoopUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopUtils.cpp
@@ -1790,6 +1790,10 @@ static bool hasHardUserWithinLoop(const Loop *L, const Instruction *I) {
   return false;
 }
 
+// Exit values of an instruction's loop-variant operands, indexed by operand
+// number.
+using OperandExitValueList = SmallVector<std::pair<unsigned, SCEVUse>, 2>;
+
 // Collect information about PHI nodes which can be transformed in
 // rewriteLoopExitValues.
 struct RewritePhi {
@@ -1798,11 +1802,14 @@ struct RewritePhi {
   SCEVUse ExpansionSCEV;     // The SCEV of the incoming value we are rewriting.
   Instruction *ExpansionPoint; // Where we'd like to expand that SCEV?
   bool HighCost;               // Is this expansion a high-cost?
+  // If SCEV cannot model the incoming instruction, use these values to clone
+  // the instruction with its loop-variant operands rewritten.
+  OperandExitValueList OperandExitValues;
 
   RewritePhi(PHINode *P, unsigned I, SCEVUse Val, Instruction *ExpansionPt,
-             bool H)
+             bool H, OperandExitValueList OperandExitValues)
       : PN(P), Ith(I), ExpansionSCEV(Val), ExpansionPoint(ExpansionPt),
-        HighCost(H) {}
+        HighCost(H), OperandExitValues(std::move(OperandExitValues)) {}
 };
 
 // Check whether it is possible to delete the loop after rewriting exit
@@ -1873,6 +1880,29 @@ static bool checkIsIndPhi(PHINode *Phi, Loop *L, ScalarEvolution *SE,
   return InductionDescriptor::isInductionPHI(Phi, L, SE, ID);
 }
 
+static bool collectOperandExitValues(Instruction *Inst, Loop *L,
+                                     BasicBlock *Preheader, ScalarEvolution *SE,
+                                     SCEVExpander &Rewriter,
+                                     OperandExitValueList &ExitValues) {
+  if (!Preheader || !isa<ICmpInst>(Inst))
+    return false;
+
+  for (const auto &[Idx, Op] : enumerate(Inst->operands())) {
+    // Operands defined outside the loop already dominate the preheader.
+    if (L->isLoopInvariant(Op))
+      continue;
+    if (!SE->isSCEVable(Op->getType()))
+      return false;
+    SCEVUse ExitValue = SE->getSCEVAtScope(Op, L->getParentLoop());
+    if (isa<SCEVCouldNotCompute>(ExitValue) ||
+        !SE->isLoopInvariant(ExitValue, L) ||
+        !Rewriter.isSafeToExpandAt(ExitValue, Preheader->getTerminator()))
+      return false;
+    ExitValues.emplace_back(Idx, ExitValue);
+  }
+  return !ExitValues.empty();
+}
+
 int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
                                 ScalarEvolution *SE,
                                 const TargetTransformInfo *TTI,
@@ -1971,6 +2001,7 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
         // expression reuse by the SCEVExpander), but resort to per-exit
         // evaluation if that fails.
         SCEVUse ExitValue = SE->getSCEVAtScope(Inst, L->getParentLoop());
+        OperandExitValueList OperandExitValues;
         if (isa<SCEVCouldNotCompute>(ExitValue) ||
             !SE->isLoopInvariant(ExitValue, L) ||
             !Rewriter.isSafeToExpand(ExitValue)) {
@@ -1979,14 +2010,15 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
           // most SCEV expressions and other recurrence types (e.g. shift
           // recurrences).  Is there existing code we can reuse?
           const SCEV *ExitCount = SE->getExitCount(L, PN->getIncomingBlock(i));
-          if (isa<SCEVCouldNotCompute>(ExitCount))
-            continue;
-          if (auto *AddRec = dyn_cast<SCEVAddRecExpr>(SE->getSCEV(Inst)))
-            if (AddRec->getLoop() == L)
-              ExitValue = AddRec->evaluateAtIteration(ExitCount, *SE);
-          if (isa<SCEVCouldNotCompute>(ExitValue) ||
-              !SE->isLoopInvariant(ExitValue, L) ||
-              !Rewriter.isSafeToExpand(ExitValue))
+          if (!isa<SCEVCouldNotCompute>(ExitCount))
+            if (auto *AddRec = dyn_cast<SCEVAddRecExpr>(SE->getSCEV(Inst)))
+              if (AddRec->getLoop() == L)
+                ExitValue = AddRec->evaluateAtIteration(ExitCount, *SE);
+          if ((isa<SCEVCouldNotCompute>(ExitValue) ||
+               !SE->isLoopInvariant(ExitValue, L) ||
+               !Rewriter.isSafeToExpand(ExitValue)) &&
+              !collectOperandExitValues(Inst, L, L->getLoopPreheader(), SE,
+                                        Rewriter, OperandExitValues))
             continue;
         }
 
@@ -2000,8 +2032,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
           continue;
 
         // Check if expansions of this SCEV would count as being high cost.
-        bool HighCost = Rewriter.isHighCostExpansion(
-            ExitValue.getPointer(), L, SCEVCheapExpansionBudget, TTI, Inst);
+        bool HighCost =
+            OperandExitValues.empty() &&
+            Rewriter.isHighCostExpansion(ExitValue.getPointer(), L,
+                                         SCEVCheapExpansionBudget, TTI, Inst);
 
         // Note that we must not perform expansions until after
         // we query *all* the costs, because if we perform temporary expansion
@@ -2013,7 +2047,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
         Instruction *InsertPt =
           (isa<PHINode>(Inst) || isa<LandingPadInst>(Inst)) ?
           &*Inst->getParent()->getFirstInsertionPt() : Inst;
-        RewritePhiSet.emplace_back(PN, i, ExitValue, InsertPt, HighCost);
+        if (!OperandExitValues.empty())
+          InsertPt = L->getLoopPreheader()->getTerminator();
+        RewritePhiSet.emplace_back(PN, i, ExitValue, InsertPt, HighCost,
+                                   std::move(OperandExitValues));
       }
     }
   }
@@ -2030,6 +2067,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
   for (const RewritePhi &Phi : RewritePhiSet) {
     PHINode *PN = Phi.PN;
 
+    // Only clone an instruction when doing so allows the loop to be deleted.
+    if (!LoopCanBeDel && !Phi.OperandExitValues.empty())
+      continue;
+
     // Only do the rewrite when the ExitValue can be expanded cheaply.
     // If LoopCanBeDel is true, rewrite exit value aggressively.
     if ((ReplaceExitValue == OnlyCheapRepl ||
@@ -2037,12 +2078,25 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
         !LoopCanBeDel && Phi.HighCost)
       continue;
 
-    Value *ExitVal = Rewriter.expandCodeFor(
-        Phi.ExpansionSCEV, Phi.PN->getType(), Phi.ExpansionPoint);
+    Instruction *Inst = cast<Instruction>(PN->getIncomingValue(Phi.Ith));
+    Value *ExitVal;
+    if (Phi.OperandExitValues.empty()) {
+      ExitVal = Rewriter.expandCodeFor(Phi.ExpansionSCEV, Phi.PN->getType(),
+                                       Phi.ExpansionPoint);
+    } else {
+      Instruction *Clone = Inst->clone();
+      Clone->setName(Inst->getName() + ".exit");
+      for (auto [Idx, ExitSCEV] : Phi.OperandExitValues)
+        Clone->setOperand(Idx, Rewriter.expandCodeFor(
+                                   ExitSCEV, Inst->getOperand(Idx)->getType(),
+                                   Phi.ExpansionPoint));
+      Clone->insertBefore(Phi.ExpansionPoint->getIterator());
+      ExitVal = Clone;
+    }
 
     LLVM_DEBUG(dbgs() << "rewriteLoopExitValues: AfterLoopVal = " << *ExitVal
                       << '\n'
-                      << "  LoopVal = " << *(Phi.ExpansionPoint) << "\n");
+                      << "  LoopVal = " << *Inst << "\n");
 
 #ifndef NDEBUG
     // If we reuse an instruction from a loop which is neither L nor one of
@@ -2055,7 +2109,6 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
 #endif
 
     NumReplaced++;
-    Instruction *Inst = cast<Instruction>(PN->getIncomingValue(Phi.Ith));
     PN->setIncomingValue(Phi.Ith, ExitVal);
     // It's necessary to tell ScalarEvolution about this explicitly so that
     // it can walk the def-use list and forget all SCEVs, as it may not be
diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index 09e8765610f69..d9677d69e48d5 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -309,17 +309,15 @@ define i1 @rewrite_computable_icmp(i32 %start, i32 %limit) {
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
-; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[IV]], 42
-; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP_EXIT]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT:    br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
 ; CHECK:       body:
-; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
 ; CHECK-NEXT:    br label [[LOOP]]
 ; CHECK:       exit:
+; CHECK-NEXT:    [[TMP0:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[START:%.*]], -42
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[TMP0]], i32 [[TMP1]])
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[START]], [[UMIN]]
+; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
 ; CHECK-NEXT:    ret i1 [[CMP_EXIT]]
 ;
 entry:
@@ -348,19 +346,17 @@ define i1 @rewrite_multiple_icmps(i32 %start, i32 %rhs.start, i32 %limit) {
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
-; CHECK-NEXT:    [[TMP2:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = phi i32 [ [[RHS_START:%.*]], [[ENTRY]] ], [ [[RHS_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
-; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP0_EXIT]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT:    br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
 ; CHECK:       body:
-; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[TMP2]], -1
-; CHECK-NEXT:    [[RHS_NEXT]] = add i32 [[TMP4]], 2
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
 ; CHECK-NEXT:    br label [[LOOP]]
 ; CHECK:       exit:
+; CHECK-NEXT:    [[TMP0:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[START:%.*]], -42
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[TMP0]], i32 [[TMP1]])
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[START]], [[UMIN]]
+; CHECK-NEXT:    [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
+; CHECK-NEXT:    [[TMP3:%.*]] = shl i32 [[UMIN]], 1
+; CHECK-NEXT:    [[TMP4:%.*]] = add i32 [[RHS_START:%.*]], [[TMP3]]
 ; CHECK-NEXT:    [[CMP1_EXIT:%.*]] = icmp sgt i32 [[TMP2]], [[TMP4]]
 ; CHECK-NEXT:    [[RESULT:%.*]] = and i1 [[CMP0_EXIT]], [[CMP1_EXIT]]
 ; CHECK-NEXT:    ret i1 [[RESULT]]
@@ -462,17 +458,18 @@ define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
-; CHECK-NEXT:    [[PTR:%.*]] = phi ptr [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[PTR_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[CMP:%.*]] = icmp ne ptr [[PTR]], [[END:%.*]]
-; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT:    br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
 ; CHECK:       body:
-; CHECK-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 1
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
 ; CHECK-NEXT:    br label [[LOOP]]
 ; CHECK:       exit:
+; CHECK-NEXT:    [[START2:%.*]] = ptrtoaddr ptr [[START:%.*]] to i64
+; CHECK-NEXT:    [[END1:%.*]] = ptrtoaddr ptr [[END:%.*]] to i64
+; CHECK-NEXT:    [[LIMIT_FR:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT:    [[TMP0:%.*]] = zext i32 [[LIMIT_FR]] to i64
+; CHECK-NEXT:    [[TMP1:%.*]] = sub i64 [[END1]], [[START2]]
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[START]], i64 [[UMIN]]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne ptr [[SCEVGEP]], [[END]]
 ; CHECK-NEXT:    ret i1 [[CMP]]
 ;
 entry:
diff --git a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
index f91972263d862..86f128debdb31 100644
--- a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
+++ b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
@@ -13,35 +13,15 @@ define i1 @rewrite_loop_exit_compare(i32 %length, i64 %limit) vscale_range(2, 10
 ; CHECK-NEXT:    [[CONTINUE3:%.*]] = and i1 [[NONZERO1]], [[INRANGE2]]
 ; CHECK-NEXT:    br i1 [[CONTINUE3]], label %[[BODY_PREHEADER:.*]], label %[[EXIT:.*]]
 ; CHECK:       [[BODY_PREHEADER]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
 ; CHECK-NEXT:    [[TMP0:%.*]] = add i32 [[LENGTH]], -1
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
 ; CHECK-NEXT:    [[TMP2:%.*]] = zext i32 [[TMP0]] to i64
 ; CHECK-NEXT:    [[UMIN:%.*]] = tail call i64 @llvm.umin.i64(i64 [[TMP1]], i64 [[TMP2]])
-; CHECK-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[UMIN]], 1
-; CHECK-NEXT:    [[TMP5:%.*]] = tail call <vscale x 16 x i32> @llvm.stepvector.nxv16i32()
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[LENGTH]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP6:%.*]] = sub nsw <vscale x 16 x i32> [[BROADCAST_SPLAT]], [[TMP5]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 16 x i32> [ [[TMP6]], %[[BODY_PREHEADER]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[AVL:%.*]] = phi i64 [ [[TMP4]], %[[BODY_PREHEADER]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP7:%.*]] = tail call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 16, i1 true)
-; CHECK-NEXT:    [[TMP8:%.*]] = zext i32 [[TMP7]] to i64
-; CHECK-NEXT:    [[TMP9:%.*]] = sub nsw i32 0, [[TMP7]]
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT9:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[TMP9]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT10:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT9]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-NEXT:    [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP8]]
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nsw <vscale x 16 x i32> [[VEC_IND]], [[BROADCAST_SPLAT10]]
-; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
-; CHECK-NEXT:    br i1 [[TMP10]], label %[[EXIT_LOOPEXIT:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK:       [[EXIT_LOOPEXIT]]:
-; CHECK-NEXT:    [[TMP11:%.*]] = add nsw i64 [[TMP8]], -1
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <vscale x 16 x i32> [[VEC_IND]], i64 [[TMP11]]
-; CHECK-NEXT:    [[TMP13:%.*]] = icmp ne i32 [[TMP12]], 1
+; CHECK-NEXT:    [[TMP3:%.*]] = trunc nuw i64 [[UMIN]] to i32
+; CHECK-NEXT:    [[NONZERO_EXIT:%.*]] = icmp ne i32 [[TMP0]], [[TMP3]]
 ; CHECK-NEXT:    br label %[[EXIT]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[TMP13]], %[[EXIT_LOOPEXIT]] ]
+; CHECK-NEXT:    [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[NONZERO_EXIT]], %[[BODY_PREHEADER]] ]
 ; CHECK-NEXT:    ret i1 [[NONZERO_LCSSA]]
 ;
 entry:
@@ -63,8 +43,3 @@ body:
 exit:
   ret i1 %nonzero
 }
-;.
-; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
-; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
-; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
-;.

>From b8b261472315495836581cc87ce168210f1dfe8f Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 18:43:27 +0800
Subject: [PATCH 3/6] [IndVars] Preserve potentially infinite subloops

Only rebuild loop-exit comparisons when LoopDeletion's progress checks
allow the loop and its subloops to be removed. This avoids making a
potentially infinite subloop unreachable.

Assisted-by: TRAE CLI (GPT-5)
---
 llvm/lib/Transforms/Utils/LoopUtils.cpp       |  25 +++-
 .../IndVarSimplify/rewrite-loop-exit-value.ll | 138 ++++++++++++++++--
 2 files changed, 147 insertions(+), 16 deletions(-)

diff --git a/llvm/lib/Transforms/Utils/LoopUtils.cpp b/llvm/lib/Transforms/Utils/LoopUtils.cpp
index 5c1a6fbaebbdd..1218098369522 100644
--- a/llvm/lib/Transforms/Utils/LoopUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopUtils.cpp
@@ -19,11 +19,13 @@
 #include "llvm/ADT/SmallVector.h"
 #include "llvm/Analysis/AliasAnalysis.h"
 #include "llvm/Analysis/BasicAliasAnalysis.h"
+#include "llvm/Analysis/CFG.h"
 #include "llvm/Analysis/DomTreeUpdater.h"
 #include "llvm/Analysis/GlobalsModRef.h"
 #include "llvm/Analysis/InstSimplifyFolder.h"
 #include "llvm/Analysis/LoopAccessAnalysis.h"
 #include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/LoopIterator.h"
 #include "llvm/Analysis/LoopPass.h"
 #include "llvm/Analysis/MemorySSA.h"
 #include "llvm/Analysis/MemorySSAUpdater.h"
@@ -1815,7 +1817,8 @@ struct RewritePhi {
 // Check whether it is possible to delete the loop after rewriting exit
 // value. If it is possible, ignore ReplaceExitValue and do rewriting
 // aggressively.
-static bool canLoopBeDeleted(Loop *L, SmallVector<RewritePhi, 8> &RewritePhiSet) {
+static bool canLoopBeDeleted(Loop *L, SmallVector<RewritePhi, 8> &RewritePhiSet,
+                             ScalarEvolution *SE, LoopInfo *LI) {
   BasicBlock *Preheader = L->getLoopPreheader();
   // If there is no preheader, the loop will not be deleted.
   if (!Preheader)
@@ -1863,6 +1866,24 @@ static bool canLoopBeDeleted(Loop *L, SmallVector<RewritePhi, 8> &RewritePhiSet)
         }))
       return false;
 
+  if (L->getHeader()->getParent()->mustProgress())
+    return true;
+
+  LoopBlocksRPO RPOT(L);
+  RPOT.perform(LI);
+  if (containsIrreducibleCFG<const BasicBlock *>(RPOT, *LI))
+    return false;
+
+  SmallVector<Loop *, 8> WorkList;
+  WorkList.push_back(L);
+  while (!WorkList.empty()) {
+    Loop *Current = WorkList.pop_back_val();
+    if (hasMustProgress(Current))
+      continue;
+    if (isa<SCEVCouldNotCompute>(SE->getConstantMaxBackedgeTakenCount(Current)))
+      return false;
+    WorkList.append(Current->begin(), Current->end());
+  }
   return true;
 }
 
@@ -2060,7 +2081,7 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
   // calculate the cost of other SCEV's after expanding SCEV 'A', thus
   // potentially giving cost bonus to those other SCEV's?
 
-  bool LoopCanBeDel = canLoopBeDeleted(L, RewritePhiSet);
+  bool LoopCanBeDel = canLoopBeDeleted(L, RewritePhiSet, SE, LI);
   int NumReplaced = 0;
 
   // Transformation.
diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index d9677d69e48d5..afa4a34b28f9d 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -4,6 +4,7 @@
 ;; Test that loop's exit value is rewritten to its initial
 ;; value from loop preheader
 define i32 @test1(ptr %var) {
+;
 ; CHECK-LABEL: @test1(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[COND:%.*]] = icmp eq ptr [[VAR:%.*]], null
@@ -34,6 +35,7 @@ exit:
 ;; Test that we can not rewrite loop exit value if it's not
 ;; a phi node (%indvar is an add instruction in this test).
 define i32 @test2(ptr %var) {
+;
 ; CHECK-LABEL: @test2(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[COND:%.*]] = icmp eq ptr [[VAR:%.*]], null
@@ -61,6 +63,7 @@ exit:
 ;; Test that we can not rewrite loop exit value if the condition
 ;; is not in loop header.
 define i32 @test3(ptr %var) {
+;
 ; CHECK-LABEL: @test3(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[COND1:%.*]] = icmp eq ptr [[VAR:%.*]], null
@@ -97,6 +100,7 @@ exit:
 
 ; Multiple exits dominating latch
 define i32 @test4(i1 %cond1, i1 %cond2) {
+;
 ; CHECK-LABEL: @test4(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[HEADER:%.*]]
@@ -124,6 +128,7 @@ exit:
 
 ; A conditionally executed exit.
 define i32 @test5(ptr %addr, i1 %cond2) {
+;
 ; CHECK-LABEL: @test5(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[HEADER:%.*]]
@@ -159,6 +164,7 @@ exit:
 }
 
 define i16 @pr57336(i16 %end, i16 %m) mustprogress {
+;
 ; CHECK-LABEL: @pr57336(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[FOR_BODY:%.*]]
@@ -198,6 +204,7 @@ crit_edge:
 }
 
 define i32 @vscale_slt_with_vp_umin(ptr nocapture %A, i32 %n) mustprogress vscale_range(2,1024) {
+;
 ; CHECK-LABEL: @vscale_slt_with_vp_umin(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[VSCALE:%.*]] = call i32 @llvm.vscale.i32()
@@ -218,12 +225,12 @@ define i32 @vscale_slt_with_vp_umin(ptr nocapture %A, i32 %n) mustprogress vscal
 ; CHECK-NEXT:    br i1 [[CMP]], label [[FOR_BODY]], label [[FOR_END:%.*]]
 ; CHECK:       for.end:
 ; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i32 [[N]], -1
-; CHECK-NEXT:    [[TMP5:%.*]] = call range(i32 2, 33) i32 @llvm.cttz.i32(i32 [[VF]], i1 true)
-; CHECK-NEXT:    [[TMP1:%.*]] = lshr i32 [[TMP0]], [[TMP5]]
-; CHECK-NEXT:    [[TMP2:%.*]] = mul i32 [[TMP1]], [[VSCALE]]
-; CHECK-NEXT:    [[TMP3:%.*]] = shl i32 [[TMP2]], 2
-; CHECK-NEXT:    [[TMP4:%.*]] = sub i32 [[N]], [[TMP3]]
-; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[VF]], i32 [[TMP4]])
+; CHECK-NEXT:    [[TMP1:%.*]] = call range(i32 2, 33) i32 @llvm.cttz.i32(i32 [[VF]], i1 true)
+; CHECK-NEXT:    [[TMP2:%.*]] = lshr i32 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = mul i32 [[TMP2]], [[VSCALE]]
+; CHECK-NEXT:    [[TMP4:%.*]] = shl i32 [[TMP3]], 2
+; CHECK-NEXT:    [[TMP5:%.*]] = sub i32 [[N]], [[TMP4]]
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[VF]], i32 [[TMP5]])
 ; CHECK-NEXT:    ret i32 [[UMIN]]
 ;
 entry:
@@ -251,6 +258,7 @@ for.end:
 }
 
 define i32 @vscale_slt_with_vp_umin2(ptr nocapture %A, i32 %n) mustprogress vscale_range(2,1024) {
+;
 ; CHECK-LABEL: @vscale_slt_with_vp_umin2(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[VSCALE:%.*]] = call i32 @llvm.vscale.i32()
@@ -271,12 +279,12 @@ define i32 @vscale_slt_with_vp_umin2(ptr nocapture %A, i32 %n) mustprogress vsca
 ; CHECK-NEXT:    br i1 [[CMP]], label [[FOR_BODY]], label [[FOR_END:%.*]]
 ; CHECK:       for.end:
 ; CHECK-NEXT:    [[TMP0:%.*]] = add i32 [[N]], -1
-; CHECK-NEXT:    [[TMP5:%.*]] = call range(i32 2, 33) i32 @llvm.cttz.i32(i32 [[VF]], i1 true)
-; CHECK-NEXT:    [[TMP1:%.*]] = lshr i32 [[TMP0]], [[TMP5]]
-; CHECK-NEXT:    [[TMP2:%.*]] = mul i32 [[TMP1]], [[VSCALE]]
-; CHECK-NEXT:    [[TMP3:%.*]] = shl i32 [[TMP2]], 2
-; CHECK-NEXT:    [[TMP4:%.*]] = sub i32 [[N]], [[TMP3]]
-; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[VF]], i32 [[TMP4]])
+; CHECK-NEXT:    [[TMP1:%.*]] = call range(i32 2, 33) i32 @llvm.cttz.i32(i32 [[VF]], i1 true)
+; CHECK-NEXT:    [[TMP2:%.*]] = lshr i32 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = mul i32 [[TMP2]], [[VSCALE]]
+; CHECK-NEXT:    [[TMP4:%.*]] = shl i32 [[TMP3]], 2
+; CHECK-NEXT:    [[TMP5:%.*]] = sub i32 [[N]], [[TMP4]]
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[VF]], i32 [[TMP5]])
 ; CHECK-NEXT:    ret i32 [[UMIN]]
 ;
 entry:
@@ -305,6 +313,7 @@ for.end:
 
 ; Rewrite a comparison by expanding its operands at the loop exit.
 define i1 @rewrite_computable_icmp(i32 %start, i32 %limit) {
+;
 ; CHECK-LABEL: @rewrite_computable_icmp(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -342,6 +351,7 @@ exit:
 
 ; Rewrite multiple comparisons when doing so makes all live-outs invariant.
 define i1 @rewrite_multiple_icmps(i32 %start, i32 %rhs.start, i32 %limit) {
+;
 ; CHECK-LABEL: @rewrite_multiple_icmps(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -387,6 +397,7 @@ exit:
 
 ; Do not rewrite a comparison if SCEV cannot compute an operand's exit value.
 define i1 @do_not_rewrite_uncomputable_icmp(i1 %c, i32 %start) {
+;
 ; CHECK-LABEL: @do_not_rewrite_uncomputable_icmp(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -415,6 +426,7 @@ exit:
 
 ; Do not rebuild a comparison unless doing so makes the loop deletable.
 define i1 @do_not_rewrite_icmp_in_live_loop(i32 %start, i32 %limit) {
+;
 ; CHECK-LABEL: @do_not_rewrite_icmp_in_live_loop(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -452,8 +464,103 @@ exit:
   ret i1 %cmp
 }
 
+; Do not rewrite a comparison if the loop contains a potentially infinite
+; subloop. Rewriting it would make the subloop unreachable and change whether
+; the function terminates.
+define i1 @do_not_rewrite_icmp_with_infinite_subloop(i32 %start, i32 %limit, i1 %c) {
+; CHECK-LABEL: @do_not_rewrite_icmp_with_infinite_subloop(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[OUTER_HEADER:%.*]]
+; CHECK:       outer.header:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[OUTER_LATCH:%.*]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[OUTER_LATCH]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i32 [[IV]], 42
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT:    br i1 [[CONTINUE]], label [[INNER_PREHEADER:%.*]], label [[EXIT:%.*]]
+; CHECK:       inner.preheader:
+; CHECK-NEXT:    br label [[INNER:%.*]]
+; CHECK:       inner:
+; CHECK-NEXT:    br i1 [[C:%.*]], label [[INNER]], label [[OUTER_LATCH]]
+; CHECK:       outer.latch:
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[OUTER_HEADER]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %outer.latch ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %outer.latch ]
+  %cmp = icmp ne i32 %iv, 42
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp, i1 %inrange, i1 false
+  br i1 %continue, label %inner, label %exit
+
+inner:
+  br i1 %c, label %inner, label %outer.latch
+
+outer.latch:
+  %iv.next = add nsw i32 %iv, -1
+  %index.next = add nuw i32 %index, 1
+  br label %outer.header
+
+exit:
+  ret i1 %cmp
+}
+
+; A mustprogress outer loop allows the same rewrite even if a subloop has an
+; unknown trip count.
+define i1 @rewrite_icmp_with_mustprogress_outer_loop(i32 %start, i32 %limit, i1 %c) {
+;
+; CHECK-LABEL: @rewrite_icmp_with_mustprogress_outer_loop(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[OUTER_HEADER:%.*]]
+; CHECK:       outer.header:
+; CHECK-NEXT:    br i1 false, label [[INNER_PREHEADER:%.*]], label [[EXIT:%.*]]
+; CHECK:       inner.preheader:
+; CHECK-NEXT:    br label [[INNER:%.*]]
+; CHECK:       inner:
+; CHECK-NEXT:    br i1 [[C:%.*]], label [[INNER]], label [[OUTER_LATCH:%.*]]
+; CHECK:       outer.latch:
+; CHECK-NEXT:    br label [[OUTER_HEADER]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       exit:
+; CHECK-NEXT:    [[TMP0:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[START:%.*]], -42
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[TMP0]], i32 [[TMP1]])
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[START]], [[UMIN]]
+; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
+; CHECK-NEXT:    ret i1 [[CMP_EXIT]]
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %outer.latch ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %outer.latch ]
+  %cmp = icmp ne i32 %iv, 42
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp, i1 %inrange, i1 false
+  br i1 %continue, label %inner, label %exit
+
+inner:
+  br i1 %c, label %inner, label %outer.latch
+
+outer.latch:
+  %iv.next = add nsw i32 %iv, -1
+  %index.next = add nuw i32 %index, 1
+  br label %outer.header, !llvm.loop !0
+
+exit:
+  ret i1 %cmp
+}
+
 ; Rewrite a pointer comparison when its operand exit values are computable.
 define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
+;
 ; CHECK-LABEL: @rewrite_pointer_icmp(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -469,8 +576,8 @@ define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
 ; CHECK-NEXT:    [[TMP1:%.*]] = sub i64 [[END1]], [[START2]]
 ; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 [[TMP1]])
 ; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[START]], i64 [[UMIN]]
-; CHECK-NEXT:    [[CMP:%.*]] = icmp ne ptr [[SCEVGEP]], [[END]]
-; CHECK-NEXT:    ret i1 [[CMP]]
+; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne ptr [[SCEVGEP]], [[END]]
+; CHECK-NEXT:    ret i1 [[CMP_EXIT]]
 ;
 entry:
   br label %loop
@@ -493,3 +600,6 @@ exit:
 }
 
 declare void @use.i1(i1)
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.mustprogress"}

>From a88bbc77f7e3b062c864b631750663e506a43eff Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Fri, 18 Sep 2026 12:13:44 +0800
Subject: [PATCH 4/6] [IndVars] Refine loop-exit value rewriting

Represent direct SCEV and operand-based expansion as mutually exclusive
alternatives. Run progress checks only when operand-based candidates
exist, and remove stray test separators.

Assisted-by: TRAE CLI (GPT-5)
---
 llvm/lib/Transforms/Utils/LoopUtils.cpp       | 66 ++++++++++++-------
 .../IndVarSimplify/rewrite-loop-exit-value.ll | 14 ----
 2 files changed, 41 insertions(+), 39 deletions(-)

diff --git a/llvm/lib/Transforms/Utils/LoopUtils.cpp b/llvm/lib/Transforms/Utils/LoopUtils.cpp
index 1218098369522..fa9020e954592 100644
--- a/llvm/lib/Transforms/Utils/LoopUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopUtils.cpp
@@ -48,6 +48,7 @@
 #include "llvm/Transforms/Utils/BasicBlockUtils.h"
 #include "llvm/Transforms/Utils/Local.h"
 #include "llvm/Transforms/Utils/ScalarEvolutionExpander.h"
+#include <variant>
 
 using namespace llvm;
 using namespace llvm::PatternMatch;
@@ -1795,30 +1796,28 @@ static bool hasHardUserWithinLoop(const Loop *L, const Instruction *I) {
 // Exit values of an instruction's loop-variant operands, indexed by operand
 // number.
 using OperandExitValueList = SmallVector<std::pair<unsigned, SCEVUse>, 2>;
+using ExpansionValue = std::variant<SCEVUse, OperandExitValueList>;
 
 // Collect information about PHI nodes which can be transformed in
 // rewriteLoopExitValues.
 struct RewritePhi {
-  PHINode *PN;               // For which PHI node is this replacement?
-  unsigned Ith;              // For which incoming value?
-  SCEVUse ExpansionSCEV;     // The SCEV of the incoming value we are rewriting.
-  Instruction *ExpansionPoint; // Where we'd like to expand that SCEV?
-  bool HighCost;               // Is this expansion a high-cost?
-  // If SCEV cannot model the incoming instruction, use these values to clone
-  // the instruction with its loop-variant operands rewritten.
-  OperandExitValueList OperandExitValues;
-
-  RewritePhi(PHINode *P, unsigned I, SCEVUse Val, Instruction *ExpansionPt,
-             bool H, OperandExitValueList OperandExitValues)
-      : PN(P), Ith(I), ExpansionSCEV(Val), ExpansionPoint(ExpansionPt),
-        HighCost(H), OperandExitValues(std::move(OperandExitValues)) {}
+  PHINode *PN;                  // For which PHI node is this replacement?
+  unsigned Ith;                 // For which incoming value?
+  ExpansionValue ValueToExpand; // The incoming value or its varying operands.
+  Instruction *ExpansionPoint;  // Where we'd like to expand that SCEV?
+  bool HighCost;                // Is this expansion a high-cost?
+
+  RewritePhi(PHINode *P, unsigned I, ExpansionValue Val,
+             Instruction *ExpansionPt, bool H)
+      : PN(P), Ith(I), ValueToExpand(std::move(Val)),
+        ExpansionPoint(ExpansionPt), HighCost(H) {}
 };
 
 // Check whether it is possible to delete the loop after rewriting exit
 // value. If it is possible, ignore ReplaceExitValue and do rewriting
 // aggressively.
-static bool canLoopBeDeleted(Loop *L, SmallVector<RewritePhi, 8> &RewritePhiSet,
-                             ScalarEvolution *SE, LoopInfo *LI) {
+static bool canLoopBeDeleted(Loop *L,
+                             SmallVector<RewritePhi, 8> &RewritePhiSet) {
   BasicBlock *Preheader = L->getLoopPreheader();
   // If there is no preheader, the loop will not be deleted.
   if (!Preheader)
@@ -1866,6 +1865,13 @@ static bool canLoopBeDeleted(Loop *L, SmallVector<RewritePhi, 8> &RewritePhiSet,
         }))
       return false;
 
+  return true;
+}
+
+// Rebuilding an exit value can make a loop newly deletable. Ensure deleting it
+// would not remove a potentially infinite loop or subloop.
+static bool loopDeletionPreservesProgress(Loop *L, ScalarEvolution *SE,
+                                          LoopInfo *LI) {
   if (L->getHeader()->getParent()->mustProgress())
     return true;
 
@@ -1874,8 +1880,7 @@ static bool canLoopBeDeleted(Loop *L, SmallVector<RewritePhi, 8> &RewritePhiSet,
   if (containsIrreducibleCFG<const BasicBlock *>(RPOT, *LI))
     return false;
 
-  SmallVector<Loop *, 8> WorkList;
-  WorkList.push_back(L);
+  SmallVector<Loop *, 8> WorkList(1, L);
   while (!WorkList.empty()) {
     Loop *Current = WorkList.pop_back_val();
     if (hasMustProgress(Current))
@@ -2070,8 +2075,11 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
           &*Inst->getParent()->getFirstInsertionPt() : Inst;
         if (!OperandExitValues.empty())
           InsertPt = L->getLoopPreheader()->getTerminator();
-        RewritePhiSet.emplace_back(PN, i, ExitValue, InsertPt, HighCost,
-                                   std::move(OperandExitValues));
+        if (OperandExitValues.empty())
+          RewritePhiSet.emplace_back(PN, i, ExitValue, InsertPt, HighCost);
+        else
+          RewritePhiSet.emplace_back(PN, i, std::move(OperandExitValues),
+                                     InsertPt, false);
       }
     }
   }
@@ -2081,7 +2089,13 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
   // calculate the cost of other SCEV's after expanding SCEV 'A', thus
   // potentially giving cost bonus to those other SCEV's?
 
-  bool LoopCanBeDel = canLoopBeDeleted(L, RewritePhiSet, SE, LI);
+  bool HasOperandExitValues =
+      llvm::any_of(RewritePhiSet, [](const RewritePhi &Phi) {
+        return std::holds_alternative<OperandExitValueList>(Phi.ValueToExpand);
+      });
+  bool LoopCanBeDel =
+      canLoopBeDeleted(L, RewritePhiSet) &&
+      (!HasOperandExitValues || loopDeletionPreservesProgress(L, SE, LI));
   int NumReplaced = 0;
 
   // Transformation.
@@ -2089,7 +2103,9 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
     PHINode *PN = Phi.PN;
 
     // Only clone an instruction when doing so allows the loop to be deleted.
-    if (!LoopCanBeDel && !Phi.OperandExitValues.empty())
+    const auto *OperandExitValues =
+        std::get_if<OperandExitValueList>(&Phi.ValueToExpand);
+    if (OperandExitValues && !LoopCanBeDel)
       continue;
 
     // Only do the rewrite when the ExitValue can be expanded cheaply.
@@ -2101,13 +2117,13 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
 
     Instruction *Inst = cast<Instruction>(PN->getIncomingValue(Phi.Ith));
     Value *ExitVal;
-    if (Phi.OperandExitValues.empty()) {
-      ExitVal = Rewriter.expandCodeFor(Phi.ExpansionSCEV, Phi.PN->getType(),
-                                       Phi.ExpansionPoint);
+    if (!OperandExitValues) {
+      ExitVal = Rewriter.expandCodeFor(std::get<SCEVUse>(Phi.ValueToExpand),
+                                       Phi.PN->getType(), Phi.ExpansionPoint);
     } else {
       Instruction *Clone = Inst->clone();
       Clone->setName(Inst->getName() + ".exit");
-      for (auto [Idx, ExitSCEV] : Phi.OperandExitValues)
+      for (auto [Idx, ExitSCEV] : *OperandExitValues)
         Clone->setOperand(Idx, Rewriter.expandCodeFor(
                                    ExitSCEV, Inst->getOperand(Idx)->getType(),
                                    Phi.ExpansionPoint));
diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index afa4a34b28f9d..02720dd45a909 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -4,7 +4,6 @@
 ;; Test that loop's exit value is rewritten to its initial
 ;; value from loop preheader
 define i32 @test1(ptr %var) {
-;
 ; CHECK-LABEL: @test1(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[COND:%.*]] = icmp eq ptr [[VAR:%.*]], null
@@ -35,7 +34,6 @@ exit:
 ;; Test that we can not rewrite loop exit value if it's not
 ;; a phi node (%indvar is an add instruction in this test).
 define i32 @test2(ptr %var) {
-;
 ; CHECK-LABEL: @test2(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[COND:%.*]] = icmp eq ptr [[VAR:%.*]], null
@@ -63,7 +61,6 @@ exit:
 ;; Test that we can not rewrite loop exit value if the condition
 ;; is not in loop header.
 define i32 @test3(ptr %var) {
-;
 ; CHECK-LABEL: @test3(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[COND1:%.*]] = icmp eq ptr [[VAR:%.*]], null
@@ -100,7 +97,6 @@ exit:
 
 ; Multiple exits dominating latch
 define i32 @test4(i1 %cond1, i1 %cond2) {
-;
 ; CHECK-LABEL: @test4(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[HEADER:%.*]]
@@ -128,7 +124,6 @@ exit:
 
 ; A conditionally executed exit.
 define i32 @test5(ptr %addr, i1 %cond2) {
-;
 ; CHECK-LABEL: @test5(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[HEADER:%.*]]
@@ -164,7 +159,6 @@ exit:
 }
 
 define i16 @pr57336(i16 %end, i16 %m) mustprogress {
-;
 ; CHECK-LABEL: @pr57336(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[FOR_BODY:%.*]]
@@ -204,7 +198,6 @@ crit_edge:
 }
 
 define i32 @vscale_slt_with_vp_umin(ptr nocapture %A, i32 %n) mustprogress vscale_range(2,1024) {
-;
 ; CHECK-LABEL: @vscale_slt_with_vp_umin(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[VSCALE:%.*]] = call i32 @llvm.vscale.i32()
@@ -258,7 +251,6 @@ for.end:
 }
 
 define i32 @vscale_slt_with_vp_umin2(ptr nocapture %A, i32 %n) mustprogress vscale_range(2,1024) {
-;
 ; CHECK-LABEL: @vscale_slt_with_vp_umin2(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[VSCALE:%.*]] = call i32 @llvm.vscale.i32()
@@ -313,7 +305,6 @@ for.end:
 
 ; Rewrite a comparison by expanding its operands at the loop exit.
 define i1 @rewrite_computable_icmp(i32 %start, i32 %limit) {
-;
 ; CHECK-LABEL: @rewrite_computable_icmp(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -351,7 +342,6 @@ exit:
 
 ; Rewrite multiple comparisons when doing so makes all live-outs invariant.
 define i1 @rewrite_multiple_icmps(i32 %start, i32 %rhs.start, i32 %limit) {
-;
 ; CHECK-LABEL: @rewrite_multiple_icmps(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -397,7 +387,6 @@ exit:
 
 ; Do not rewrite a comparison if SCEV cannot compute an operand's exit value.
 define i1 @do_not_rewrite_uncomputable_icmp(i1 %c, i32 %start) {
-;
 ; CHECK-LABEL: @do_not_rewrite_uncomputable_icmp(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -426,7 +415,6 @@ exit:
 
 ; Do not rebuild a comparison unless doing so makes the loop deletable.
 define i1 @do_not_rewrite_icmp_in_live_loop(i32 %start, i32 %limit) {
-;
 ; CHECK-LABEL: @do_not_rewrite_icmp_in_live_loop(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -515,7 +503,6 @@ exit:
 ; A mustprogress outer loop allows the same rewrite even if a subloop has an
 ; unknown trip count.
 define i1 @rewrite_icmp_with_mustprogress_outer_loop(i32 %start, i32 %limit, i1 %c) {
-;
 ; CHECK-LABEL: @rewrite_icmp_with_mustprogress_outer_loop(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[OUTER_HEADER:%.*]]
@@ -560,7 +547,6 @@ exit:
 
 ; Rewrite a pointer comparison when its operand exit values are computable.
 define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
-;
 ; CHECK-LABEL: @rewrite_pointer_icmp(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]

>From 305a8eabbf4d1a95784c67bb471c99c8a812b229 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Fri, 18 Sep 2026 17:41:30 +0800
Subject: [PATCH 5/6] [IndVars] Document an existing progress issue

Add FIXME coverage showing that SCEV exit-value rewriting can expose
an existing progress bug in predicateLoopExits. Keep the fix separate
from operand-based exit-value rewriting.

Assisted-by: TRAE CLI (GPT-5)
---
 llvm/lib/Transforms/Utils/LoopUtils.cpp       |   2 +
 ...ewrite-loop-exit-value-infinite-subloop.ll | 104 ++++++++++++++++++
 2 files changed, 106 insertions(+)
 create mode 100644 llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value-infinite-subloop.ll

diff --git a/llvm/lib/Transforms/Utils/LoopUtils.cpp b/llvm/lib/Transforms/Utils/LoopUtils.cpp
index fa9020e954592..5098e4131ab3f 100644
--- a/llvm/lib/Transforms/Utils/LoopUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopUtils.cpp
@@ -2093,6 +2093,8 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
       llvm::any_of(RewritePhiSet, [](const RewritePhi &Phi) {
         return std::holds_alternative<OperandExitValueList>(Phi.ValueToExpand);
       });
+  // FIXME: SCEV-based rewrites can also expose an existing progress bug in
+  // predicateLoopExits().
   bool LoopCanBeDel =
       canLoopBeDeleted(L, RewritePhiSet) &&
       (!HasOperandExitValues || loopDeletionPreservesProgress(L, SE, LI));
diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value-infinite-subloop.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value-infinite-subloop.ll
new file mode 100644
index 0000000000000..c89c8b0662610
--- /dev/null
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value-infinite-subloop.ll
@@ -0,0 +1,104 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes='loop(indvars)' -S < %s | FileCheck %s
+
+; FIXME: These are miscompiles. Neither @scev nor @loaded_step is marked
+; mustprogress, and neither inner loop carries llvm.loop.mustprogress, so a
+; potentially infinite inner loop must be preserved. Rewriting the outer
+; loop's exit value removes its last live-out, after which predicateLoopExits
+; folds the outer header exit to a constant and makes the inner loop
+; unreachable.
+;
+; This is not reachable from the default optimization pipelines today because
+; LoopRotate moves the outer exit test to the latch before IndVarSimplify,
+; making the inner loop execute at least once.
+
+; The inner loop is infinite when %c is true and %n is nonzero.
+define i32 @scev(i32 %n, i1 %c) {
+; CHECK-LABEL: define i32 @scev(
+; CHECK-SAME: i32 [[N:%.*]], i1 [[C:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    br i1 false, label %[[INNER_PREHEADER:.*]], label %[[EXIT:.*]]
+; CHECK:       [[INNER_PREHEADER]]:
+; CHECK-NEXT:    br label %[[INNER:.*]]
+; CHECK:       [[INNER]]:
+; CHECK-NEXT:    br i1 [[C]], label %[[INNER]], label %[[OUTER_LATCH:.*]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    br label %[[OUTER_HEADER]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[N]], i32 1)
+; CHECK-NEXT:    [[TMP0:%.*]] = sub i32 [[N]], [[UMIN]]
+; CHECK-NEXT:    [[TMP1:%.*]] = udiv i32 [[TMP0]], 3
+; CHECK-NEXT:    [[TMP2:%.*]] = add i32 [[UMIN]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = mul i32 [[TMP2]], 3
+; CHECK-NEXT:    ret i32 [[TMP3]]
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %outer.latch ]
+  %ec = icmp ult i32 %iv, %n
+  br i1 %ec, label %inner, label %exit
+
+inner:
+  br i1 %c, label %inner, label %outer.latch
+
+outer.latch:
+  %iv.next = add nuw nsw i32 %iv, 3
+  br label %outer.header
+
+exit:
+  ret i32 %iv
+}
+
+; The inner loop's trip count is unknown because its step is loaded. It is
+; infinite when the loaded step is zero and %m is nonzero.
+define i32 @loaded_step(i32 %n, i32 %m, ptr %p) {
+; CHECK-LABEL: define i32 @loaded_step(
+; CHECK-SAME: i32 [[N:%.*]], i32 [[M:%.*]], ptr [[P:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    br label %[[OUTER_HEADER:.*]]
+; CHECK:       [[OUTER_HEADER]]:
+; CHECK-NEXT:    br i1 false, label %[[INNER_PREHEADER:.*]], label %[[EXIT:.*]]
+; CHECK:       [[INNER_PREHEADER]]:
+; CHECK-NEXT:    br label %[[INNER:.*]]
+; CHECK:       [[INNER]]:
+; CHECK-NEXT:    [[J:%.*]] = phi i32 [ [[J_NEXT:%.*]], %[[INNER]] ], [ 0, %[[INNER_PREHEADER]] ]
+; CHECK-NEXT:    [[A:%.*]] = load i32, ptr [[P]], align 4
+; CHECK-NEXT:    [[J_NEXT]] = add i32 [[J]], [[A]]
+; CHECK-NEXT:    [[ICOND:%.*]] = icmp ult i32 [[J_NEXT]], [[M]]
+; CHECK-NEXT:    br i1 [[ICOND]], label %[[INNER]], label %[[OUTER_LATCH:.*]]
+; CHECK:       [[OUTER_LATCH]]:
+; CHECK-NEXT:    br label %[[OUTER_HEADER]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[N]], i32 1)
+; CHECK-NEXT:    [[TMP0:%.*]] = sub i32 [[N]], [[UMIN]]
+; CHECK-NEXT:    [[TMP1:%.*]] = udiv i32 [[TMP0]], 3
+; CHECK-NEXT:    [[TMP2:%.*]] = add i32 [[UMIN]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = mul i32 [[TMP2]], 3
+; CHECK-NEXT:    ret i32 [[TMP3]]
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %outer.latch ]
+  %ec = icmp ult i32 %iv, %n
+  br i1 %ec, label %inner, label %exit
+
+inner:
+  %j = phi i32 [ 0, %outer.header ], [ %j.next, %inner ]
+  %a = load i32, ptr %p
+  %j.next = add i32 %j, %a
+  %icond = icmp ult i32 %j.next, %m
+  br i1 %icond, label %inner, label %outer.latch
+
+outer.latch:
+  %iv.next = add nuw nsw i32 %iv, 3
+  br label %outer.header
+
+exit:
+  ret i32 %iv
+}

>From 2b4ba18e7d70c097bdd5466de9019117ed95e619 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Sun, 20 Sep 2026 15:33:13 +0800
Subject: [PATCH 6/6] [IndVars] Leave progress handling to LoopDeletion

Keep this change focused on operand-based exit-value rewriting. Leave
the existing loop progress issue to its dedicated fix.

Assisted-by: TRAE CLI (GPT-5)
---
 llvm/lib/Transforms/Utils/LoopUtils.cpp       | 36 +------
 .../IndVarSimplify/rewrite-loop-exit-value.ll | 96 -------------------
 2 files changed, 1 insertion(+), 131 deletions(-)

diff --git a/llvm/lib/Transforms/Utils/LoopUtils.cpp b/llvm/lib/Transforms/Utils/LoopUtils.cpp
index 5098e4131ab3f..90be744087bfb 100644
--- a/llvm/lib/Transforms/Utils/LoopUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopUtils.cpp
@@ -19,13 +19,11 @@
 #include "llvm/ADT/SmallVector.h"
 #include "llvm/Analysis/AliasAnalysis.h"
 #include "llvm/Analysis/BasicAliasAnalysis.h"
-#include "llvm/Analysis/CFG.h"
 #include "llvm/Analysis/DomTreeUpdater.h"
 #include "llvm/Analysis/GlobalsModRef.h"
 #include "llvm/Analysis/InstSimplifyFolder.h"
 #include "llvm/Analysis/LoopAccessAnalysis.h"
 #include "llvm/Analysis/LoopInfo.h"
-#include "llvm/Analysis/LoopIterator.h"
 #include "llvm/Analysis/LoopPass.h"
 #include "llvm/Analysis/MemorySSA.h"
 #include "llvm/Analysis/MemorySSAUpdater.h"
@@ -1868,30 +1866,6 @@ static bool canLoopBeDeleted(Loop *L,
   return true;
 }
 
-// Rebuilding an exit value can make a loop newly deletable. Ensure deleting it
-// would not remove a potentially infinite loop or subloop.
-static bool loopDeletionPreservesProgress(Loop *L, ScalarEvolution *SE,
-                                          LoopInfo *LI) {
-  if (L->getHeader()->getParent()->mustProgress())
-    return true;
-
-  LoopBlocksRPO RPOT(L);
-  RPOT.perform(LI);
-  if (containsIrreducibleCFG<const BasicBlock *>(RPOT, *LI))
-    return false;
-
-  SmallVector<Loop *, 8> WorkList(1, L);
-  while (!WorkList.empty()) {
-    Loop *Current = WorkList.pop_back_val();
-    if (hasMustProgress(Current))
-      continue;
-    if (isa<SCEVCouldNotCompute>(SE->getConstantMaxBackedgeTakenCount(Current)))
-      return false;
-    WorkList.append(Current->begin(), Current->end());
-  }
-  return true;
-}
-
 /// Checks if it is safe to call InductionDescriptor::isInductionPHI for \p Phi,
 /// and returns true if this Phi is an induction phi in the loop. When
 /// isInductionPHI returns true, \p ID will be also be set by isInductionPHI.
@@ -2089,15 +2063,7 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
   // calculate the cost of other SCEV's after expanding SCEV 'A', thus
   // potentially giving cost bonus to those other SCEV's?
 
-  bool HasOperandExitValues =
-      llvm::any_of(RewritePhiSet, [](const RewritePhi &Phi) {
-        return std::holds_alternative<OperandExitValueList>(Phi.ValueToExpand);
-      });
-  // FIXME: SCEV-based rewrites can also expose an existing progress bug in
-  // predicateLoopExits().
-  bool LoopCanBeDel =
-      canLoopBeDeleted(L, RewritePhiSet) &&
-      (!HasOperandExitValues || loopDeletionPreservesProgress(L, SE, LI));
+  bool LoopCanBeDel = canLoopBeDeleted(L, RewritePhiSet);
   int NumReplaced = 0;
 
   // Transformation.
diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index 02720dd45a909..8b999063c6907 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -452,99 +452,6 @@ exit:
   ret i1 %cmp
 }
 
-; Do not rewrite a comparison if the loop contains a potentially infinite
-; subloop. Rewriting it would make the subloop unreachable and change whether
-; the function terminates.
-define i1 @do_not_rewrite_icmp_with_infinite_subloop(i32 %start, i32 %limit, i1 %c) {
-; CHECK-LABEL: @do_not_rewrite_icmp_with_infinite_subloop(
-; CHECK-NEXT:  entry:
-; CHECK-NEXT:    br label [[OUTER_HEADER:%.*]]
-; CHECK:       outer.header:
-; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[OUTER_LATCH:%.*]] ]
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[OUTER_LATCH]] ]
-; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i32 [[IV]], 42
-; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT:    br i1 [[CONTINUE]], label [[INNER_PREHEADER:%.*]], label [[EXIT:%.*]]
-; CHECK:       inner.preheader:
-; CHECK-NEXT:    br label [[INNER:%.*]]
-; CHECK:       inner:
-; CHECK-NEXT:    br i1 [[C:%.*]], label [[INNER]], label [[OUTER_LATCH]]
-; CHECK:       outer.latch:
-; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
-; CHECK-NEXT:    br label [[OUTER_HEADER]]
-; CHECK:       exit:
-; CHECK-NEXT:    ret i1 [[CMP]]
-;
-entry:
-  br label %outer.header
-
-outer.header:
-  %iv = phi i32 [ %start, %entry ], [ %iv.next, %outer.latch ]
-  %index = phi i32 [ 0, %entry ], [ %index.next, %outer.latch ]
-  %cmp = icmp ne i32 %iv, 42
-  %inrange = icmp ult i32 %index, %limit
-  %continue = select i1 %cmp, i1 %inrange, i1 false
-  br i1 %continue, label %inner, label %exit
-
-inner:
-  br i1 %c, label %inner, label %outer.latch
-
-outer.latch:
-  %iv.next = add nsw i32 %iv, -1
-  %index.next = add nuw i32 %index, 1
-  br label %outer.header
-
-exit:
-  ret i1 %cmp
-}
-
-; A mustprogress outer loop allows the same rewrite even if a subloop has an
-; unknown trip count.
-define i1 @rewrite_icmp_with_mustprogress_outer_loop(i32 %start, i32 %limit, i1 %c) {
-; CHECK-LABEL: @rewrite_icmp_with_mustprogress_outer_loop(
-; CHECK-NEXT:  entry:
-; CHECK-NEXT:    br label [[OUTER_HEADER:%.*]]
-; CHECK:       outer.header:
-; CHECK-NEXT:    br i1 false, label [[INNER_PREHEADER:%.*]], label [[EXIT:%.*]]
-; CHECK:       inner.preheader:
-; CHECK-NEXT:    br label [[INNER:%.*]]
-; CHECK:       inner:
-; CHECK-NEXT:    br i1 [[C:%.*]], label [[INNER]], label [[OUTER_LATCH:%.*]]
-; CHECK:       outer.latch:
-; CHECK-NEXT:    br label [[OUTER_HEADER]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK:       exit:
-; CHECK-NEXT:    [[TMP0:%.*]] = freeze i32 [[LIMIT:%.*]]
-; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[START:%.*]], -42
-; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[TMP0]], i32 [[TMP1]])
-; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[START]], [[UMIN]]
-; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
-; CHECK-NEXT:    ret i1 [[CMP_EXIT]]
-;
-entry:
-  br label %outer.header
-
-outer.header:
-  %iv = phi i32 [ %start, %entry ], [ %iv.next, %outer.latch ]
-  %index = phi i32 [ 0, %entry ], [ %index.next, %outer.latch ]
-  %cmp = icmp ne i32 %iv, 42
-  %inrange = icmp ult i32 %index, %limit
-  %continue = select i1 %cmp, i1 %inrange, i1 false
-  br i1 %continue, label %inner, label %exit
-
-inner:
-  br i1 %c, label %inner, label %outer.latch
-
-outer.latch:
-  %iv.next = add nsw i32 %iv, -1
-  %index.next = add nuw i32 %index, 1
-  br label %outer.header, !llvm.loop !0
-
-exit:
-  ret i1 %cmp
-}
-
 ; Rewrite a pointer comparison when its operand exit values are computable.
 define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
 ; CHECK-LABEL: @rewrite_pointer_icmp(
@@ -586,6 +493,3 @@ exit:
 }
 
 declare void @use.i1(i1)
-
-!0 = distinct !{!0, !1}
-!1 = !{!"llvm.loop.mustprogress"}



More information about the llvm-commits mailing list