[llvm] [IndVars] Rewrite loop-exit comparisons from operands (PR #223951)

Pengcheng Wang via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 17 03:54:04 PDT 2026


https://github.com/wangpc-pp updated https://github.com/llvm/llvm-project/pull/223951

>From 31532d528e1c67110a63616b53edddab203b1737 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 12:58:40 +0800
Subject: [PATCH 1/3] [IndVars] Add tests for rewriting loop-exit comparisons

Add coverage for integer and pointer comparisons, multiple varying

operands, uncomputable exit values, and the RISC-V phase-ordering

issue.

Assisted-by: TRAE CLI (GPT-5)
---
 .../IndVarSimplify/rewrite-loop-exit-value.ll | 194 ++++++++++++++++++
 .../RISCV/rewrite-loop-exit-compare.ll        |  70 +++++++
 2 files changed, 264 insertions(+)
 create mode 100644 llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll

diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index fa47d06d859e9..09e8765610f69 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -302,3 +302,197 @@ for.body:
 for.end:
   ret i32 %VF.capped
 }
+
+; Rewrite a comparison by expanding its operands at the loop exit.
+define i1 @rewrite_computable_icmp(i32 %start, i32 %limit) {
+; CHECK-LABEL: @rewrite_computable_icmp(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[IV]], 42
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP_EXIT]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP_EXIT]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp = icmp ne i32 %iv, 42
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %iv.next = add nsw i32 %iv, -1
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %cmp
+}
+
+; Rewrite multiple comparisons when doing so makes all live-outs invariant.
+define i1 @rewrite_multiple_icmps(i32 %start, i32 %rhs.start, i32 %limit) {
+; CHECK-LABEL: @rewrite_multiple_icmps(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[TMP2:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = phi i32 [ [[RHS_START:%.*]], [[ENTRY]] ], [ [[RHS_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP0_EXIT]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[TMP2]], -1
+; CHECK-NEXT:    [[RHS_NEXT]] = add i32 [[TMP4]], 2
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    [[CMP1_EXIT:%.*]] = icmp sgt i32 [[TMP2]], [[TMP4]]
+; CHECK-NEXT:    [[RESULT:%.*]] = and i1 [[CMP0_EXIT]], [[CMP1_EXIT]]
+; CHECK-NEXT:    ret i1 [[RESULT]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+  %rhs = phi i32 [ %rhs.start, %entry ], [ %rhs.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp0 = icmp ne i32 %iv, 42
+  %cmp1 = icmp sgt i32 %iv, %rhs
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp0, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %iv.next = add nsw i32 %iv, -1
+  %rhs.next = add i32 %rhs, 2
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  %result = and i1 %cmp0, %cmp1
+  ret i1 %result
+}
+
+; Do not rewrite a comparison if SCEV cannot compute an operand's exit value.
+define i1 @do_not_rewrite_uncomputable_icmp(i1 %c, i32 %start) {
+; CHECK-LABEL: @do_not_rewrite_uncomputable_icmp(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i32 [[IV]], 0
+; CHECK-NEXT:    [[SELECTED:%.*]] = select i1 [[CMP]], i1 [[C:%.*]], i1 false
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT:    br i1 [[SELECTED]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %loop ]
+  %cmp = icmp ne i32 %iv, 0
+  %selected = select i1 %cmp, i1 %c, i1 false
+  %iv.next = add i32 %iv, 1
+  br i1 %selected, label %loop, label %exit
+
+exit:
+  ret i1 %cmp
+}
+
+; Do not rebuild a comparison unless doing so makes the loop deletable.
+define i1 @do_not_rewrite_icmp_in_live_loop(i32 %start, i32 %limit) {
+; CHECK-LABEL: @do_not_rewrite_icmp_in_live_loop(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i32 [[IV]], 42
+; CHECK-NEXT:    call void @use.i1(i1 [[CMP]])
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    br i1 [[INRANGE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp = icmp ne i32 %iv, 42
+  call void @use.i1(i1 %cmp)
+  %inrange = icmp ult i32 %index, %limit
+  br i1 %inrange, label %body, label %exit
+
+body:
+  %iv.next = add nsw i32 %iv, -1
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %cmp
+}
+
+; Rewrite a pointer comparison when its operand exit values are computable.
+define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
+; CHECK-LABEL: @rewrite_pointer_icmp(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[PTR:%.*]] = phi ptr [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[PTR_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne ptr [[PTR]], [[END:%.*]]
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %ptr = phi ptr [ %start, %entry ], [ %ptr.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp = icmp ne ptr %ptr, %end
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %ptr.next = getelementptr i8, ptr %ptr, i64 1
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %cmp
+}
+
+declare void @use.i1(i1)
diff --git a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
new file mode 100644
index 0000000000000..f91972263d862
--- /dev/null
+++ b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
@@ -0,0 +1,70 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -mtriple=riscv64 -mattr=+v -passes='default<O3>' -S %s | FileCheck %s
+
+; Rewriting the comparison from its operands' exit values allows the loop to be
+; deleted instead of vectorizing the induction and extracting its final result.
+define i1 @rewrite_loop_exit_compare(i32 %length, i64 %limit) vscale_range(2, 1024) {
+; CHECK-LABEL: define i1 @rewrite_loop_exit_compare(
+; CHECK-SAME: i32 [[LENGTH:%.*]], i64 [[LIMIT:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[LIMIT_FR:%.*]] = freeze i64 [[LIMIT]]
+; CHECK-NEXT:    [[NONZERO1:%.*]] = icmp ne i32 [[LENGTH]], 0
+; CHECK-NEXT:    [[INRANGE2:%.*]] = icmp ne i64 [[LIMIT_FR]], 0
+; CHECK-NEXT:    [[CONTINUE3:%.*]] = and i1 [[NONZERO1]], [[INRANGE2]]
+; CHECK-NEXT:    br i1 [[CONTINUE3]], label %[[BODY_PREHEADER:.*]], label %[[EXIT:.*]]
+; CHECK:       [[BODY_PREHEADER]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
+; CHECK-NEXT:    [[TMP0:%.*]] = add i32 [[LENGTH]], -1
+; CHECK-NEXT:    [[TMP2:%.*]] = zext i32 [[TMP0]] to i64
+; CHECK-NEXT:    [[UMIN:%.*]] = tail call i64 @llvm.umin.i64(i64 [[TMP1]], i64 [[TMP2]])
+; CHECK-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[UMIN]], 1
+; CHECK-NEXT:    [[TMP5:%.*]] = tail call <vscale x 16 x i32> @llvm.stepvector.nxv16i32()
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[LENGTH]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP6:%.*]] = sub nsw <vscale x 16 x i32> [[BROADCAST_SPLAT]], [[TMP5]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 16 x i32> [ [[TMP6]], %[[BODY_PREHEADER]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[AVL:%.*]] = phi i64 [ [[TMP4]], %[[BODY_PREHEADER]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP7:%.*]] = tail call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 16, i1 true)
+; CHECK-NEXT:    [[TMP8:%.*]] = zext i32 [[TMP7]] to i64
+; CHECK-NEXT:    [[TMP9:%.*]] = sub nsw i32 0, [[TMP7]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT9:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[TMP9]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT10:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT9]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-NEXT:    [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP8]]
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nsw <vscale x 16 x i32> [[VEC_IND]], [[BROADCAST_SPLAT10]]
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[EXIT_LOOPEXIT:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[EXIT_LOOPEXIT]]:
+; CHECK-NEXT:    [[TMP11:%.*]] = add nsw i64 [[TMP8]], -1
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <vscale x 16 x i32> [[VEC_IND]], i64 [[TMP11]]
+; CHECK-NEXT:    [[TMP13:%.*]] = icmp ne i32 [[TMP12]], 1
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[TMP13]], %[[EXIT_LOOPEXIT]] ]
+; CHECK-NEXT:    ret i1 [[NONZERO_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %length.addr = phi i32 [ %length, %entry ], [ %dec, %body ]
+  %index = phi i64 [ 0, %entry ], [ %index.next, %body ]
+  %nonzero = icmp ne i32 %length.addr, 0
+  %inrange = icmp ult i64 %index, %limit
+  %continue = select i1 %nonzero, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %dec = add nsw i32 %length.addr, -1
+  %index.next = add nuw i64 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %nonzero
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+;.

>From 276cb73e2b17e56fb1d11db1d16f4eaef7895992 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 13:00:15 +0800
Subject: [PATCH 2/3] [IndVars] Rewrite loop-exit comparisons from operands

When SCEV cannot model a comparison, clone it in the loop preheader

using the exit values of its varying operands. This lets exit-value

rewriting remove otherwise dead loops before vectorization.

Assisted-by: TRAE CLI (GPT-5)
---
 llvm/lib/Transforms/Utils/LoopUtils.cpp       | 87 +++++++++++++++----
 .../IndVarSimplify/rewrite-loop-exit-value.ll | 49 +++++------
 .../RISCV/rewrite-loop-exit-compare.ll        | 33 +------
 3 files changed, 97 insertions(+), 72 deletions(-)

diff --git a/llvm/lib/Transforms/Utils/LoopUtils.cpp b/llvm/lib/Transforms/Utils/LoopUtils.cpp
index a2e544801b9c4..5c1a6fbaebbdd 100644
--- a/llvm/lib/Transforms/Utils/LoopUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopUtils.cpp
@@ -1790,6 +1790,10 @@ static bool hasHardUserWithinLoop(const Loop *L, const Instruction *I) {
   return false;
 }
 
+// Exit values of an instruction's loop-variant operands, indexed by operand
+// number.
+using OperandExitValueList = SmallVector<std::pair<unsigned, SCEVUse>, 2>;
+
 // Collect information about PHI nodes which can be transformed in
 // rewriteLoopExitValues.
 struct RewritePhi {
@@ -1798,11 +1802,14 @@ struct RewritePhi {
   SCEVUse ExpansionSCEV;     // The SCEV of the incoming value we are rewriting.
   Instruction *ExpansionPoint; // Where we'd like to expand that SCEV?
   bool HighCost;               // Is this expansion a high-cost?
+  // If SCEV cannot model the incoming instruction, use these values to clone
+  // the instruction with its loop-variant operands rewritten.
+  OperandExitValueList OperandExitValues;
 
   RewritePhi(PHINode *P, unsigned I, SCEVUse Val, Instruction *ExpansionPt,
-             bool H)
+             bool H, OperandExitValueList OperandExitValues)
       : PN(P), Ith(I), ExpansionSCEV(Val), ExpansionPoint(ExpansionPt),
-        HighCost(H) {}
+        HighCost(H), OperandExitValues(std::move(OperandExitValues)) {}
 };
 
 // Check whether it is possible to delete the loop after rewriting exit
@@ -1873,6 +1880,29 @@ static bool checkIsIndPhi(PHINode *Phi, Loop *L, ScalarEvolution *SE,
   return InductionDescriptor::isInductionPHI(Phi, L, SE, ID);
 }
 
+static bool collectOperandExitValues(Instruction *Inst, Loop *L,
+                                     BasicBlock *Preheader, ScalarEvolution *SE,
+                                     SCEVExpander &Rewriter,
+                                     OperandExitValueList &ExitValues) {
+  if (!Preheader || !isa<ICmpInst>(Inst))
+    return false;
+
+  for (const auto &[Idx, Op] : enumerate(Inst->operands())) {
+    // Operands defined outside the loop already dominate the preheader.
+    if (L->isLoopInvariant(Op))
+      continue;
+    if (!SE->isSCEVable(Op->getType()))
+      return false;
+    SCEVUse ExitValue = SE->getSCEVAtScope(Op, L->getParentLoop());
+    if (isa<SCEVCouldNotCompute>(ExitValue) ||
+        !SE->isLoopInvariant(ExitValue, L) ||
+        !Rewriter.isSafeToExpandAt(ExitValue, Preheader->getTerminator()))
+      return false;
+    ExitValues.emplace_back(Idx, ExitValue);
+  }
+  return !ExitValues.empty();
+}
+
 int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
                                 ScalarEvolution *SE,
                                 const TargetTransformInfo *TTI,
@@ -1971,6 +2001,7 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
         // expression reuse by the SCEVExpander), but resort to per-exit
         // evaluation if that fails.
         SCEVUse ExitValue = SE->getSCEVAtScope(Inst, L->getParentLoop());
+        OperandExitValueList OperandExitValues;
         if (isa<SCEVCouldNotCompute>(ExitValue) ||
             !SE->isLoopInvariant(ExitValue, L) ||
             !Rewriter.isSafeToExpand(ExitValue)) {
@@ -1979,14 +2010,15 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
           // most SCEV expressions and other recurrence types (e.g. shift
           // recurrences).  Is there existing code we can reuse?
           const SCEV *ExitCount = SE->getExitCount(L, PN->getIncomingBlock(i));
-          if (isa<SCEVCouldNotCompute>(ExitCount))
-            continue;
-          if (auto *AddRec = dyn_cast<SCEVAddRecExpr>(SE->getSCEV(Inst)))
-            if (AddRec->getLoop() == L)
-              ExitValue = AddRec->evaluateAtIteration(ExitCount, *SE);
-          if (isa<SCEVCouldNotCompute>(ExitValue) ||
-              !SE->isLoopInvariant(ExitValue, L) ||
-              !Rewriter.isSafeToExpand(ExitValue))
+          if (!isa<SCEVCouldNotCompute>(ExitCount))
+            if (auto *AddRec = dyn_cast<SCEVAddRecExpr>(SE->getSCEV(Inst)))
+              if (AddRec->getLoop() == L)
+                ExitValue = AddRec->evaluateAtIteration(ExitCount, *SE);
+          if ((isa<SCEVCouldNotCompute>(ExitValue) ||
+               !SE->isLoopInvariant(ExitValue, L) ||
+               !Rewriter.isSafeToExpand(ExitValue)) &&
+              !collectOperandExitValues(Inst, L, L->getLoopPreheader(), SE,
+                                        Rewriter, OperandExitValues))
             continue;
         }
 
@@ -2000,8 +2032,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
           continue;
 
         // Check if expansions of this SCEV would count as being high cost.
-        bool HighCost = Rewriter.isHighCostExpansion(
-            ExitValue.getPointer(), L, SCEVCheapExpansionBudget, TTI, Inst);
+        bool HighCost =
+            OperandExitValues.empty() &&
+            Rewriter.isHighCostExpansion(ExitValue.getPointer(), L,
+                                         SCEVCheapExpansionBudget, TTI, Inst);
 
         // Note that we must not perform expansions until after
         // we query *all* the costs, because if we perform temporary expansion
@@ -2013,7 +2047,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
         Instruction *InsertPt =
           (isa<PHINode>(Inst) || isa<LandingPadInst>(Inst)) ?
           &*Inst->getParent()->getFirstInsertionPt() : Inst;
-        RewritePhiSet.emplace_back(PN, i, ExitValue, InsertPt, HighCost);
+        if (!OperandExitValues.empty())
+          InsertPt = L->getLoopPreheader()->getTerminator();
+        RewritePhiSet.emplace_back(PN, i, ExitValue, InsertPt, HighCost,
+                                   std::move(OperandExitValues));
       }
     }
   }
@@ -2030,6 +2067,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
   for (const RewritePhi &Phi : RewritePhiSet) {
     PHINode *PN = Phi.PN;
 
+    // Only clone an instruction when doing so allows the loop to be deleted.
+    if (!LoopCanBeDel && !Phi.OperandExitValues.empty())
+      continue;
+
     // Only do the rewrite when the ExitValue can be expanded cheaply.
     // If LoopCanBeDel is true, rewrite exit value aggressively.
     if ((ReplaceExitValue == OnlyCheapRepl ||
@@ -2037,12 +2078,25 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
         !LoopCanBeDel && Phi.HighCost)
       continue;
 
-    Value *ExitVal = Rewriter.expandCodeFor(
-        Phi.ExpansionSCEV, Phi.PN->getType(), Phi.ExpansionPoint);
+    Instruction *Inst = cast<Instruction>(PN->getIncomingValue(Phi.Ith));
+    Value *ExitVal;
+    if (Phi.OperandExitValues.empty()) {
+      ExitVal = Rewriter.expandCodeFor(Phi.ExpansionSCEV, Phi.PN->getType(),
+                                       Phi.ExpansionPoint);
+    } else {
+      Instruction *Clone = Inst->clone();
+      Clone->setName(Inst->getName() + ".exit");
+      for (auto [Idx, ExitSCEV] : Phi.OperandExitValues)
+        Clone->setOperand(Idx, Rewriter.expandCodeFor(
+                                   ExitSCEV, Inst->getOperand(Idx)->getType(),
+                                   Phi.ExpansionPoint));
+      Clone->insertBefore(Phi.ExpansionPoint->getIterator());
+      ExitVal = Clone;
+    }
 
     LLVM_DEBUG(dbgs() << "rewriteLoopExitValues: AfterLoopVal = " << *ExitVal
                       << '\n'
-                      << "  LoopVal = " << *(Phi.ExpansionPoint) << "\n");
+                      << "  LoopVal = " << *Inst << "\n");
 
 #ifndef NDEBUG
     // If we reuse an instruction from a loop which is neither L nor one of
@@ -2055,7 +2109,6 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
 #endif
 
     NumReplaced++;
-    Instruction *Inst = cast<Instruction>(PN->getIncomingValue(Phi.Ith));
     PN->setIncomingValue(Phi.Ith, ExitVal);
     // It's necessary to tell ScalarEvolution about this explicitly so that
     // it can walk the def-use list and forget all SCEVs, as it may not be
diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index 09e8765610f69..d9677d69e48d5 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -309,17 +309,15 @@ define i1 @rewrite_computable_icmp(i32 %start, i32 %limit) {
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
-; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[IV]], 42
-; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP_EXIT]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT:    br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
 ; CHECK:       body:
-; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
 ; CHECK-NEXT:    br label [[LOOP]]
 ; CHECK:       exit:
+; CHECK-NEXT:    [[TMP0:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[START:%.*]], -42
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[TMP0]], i32 [[TMP1]])
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[START]], [[UMIN]]
+; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
 ; CHECK-NEXT:    ret i1 [[CMP_EXIT]]
 ;
 entry:
@@ -348,19 +346,17 @@ define i1 @rewrite_multiple_icmps(i32 %start, i32 %rhs.start, i32 %limit) {
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
-; CHECK-NEXT:    [[TMP2:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = phi i32 [ [[RHS_START:%.*]], [[ENTRY]] ], [ [[RHS_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
-; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP0_EXIT]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT:    br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
 ; CHECK:       body:
-; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[TMP2]], -1
-; CHECK-NEXT:    [[RHS_NEXT]] = add i32 [[TMP4]], 2
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
 ; CHECK-NEXT:    br label [[LOOP]]
 ; CHECK:       exit:
+; CHECK-NEXT:    [[TMP0:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[START:%.*]], -42
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[TMP0]], i32 [[TMP1]])
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[START]], [[UMIN]]
+; CHECK-NEXT:    [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
+; CHECK-NEXT:    [[TMP3:%.*]] = shl i32 [[UMIN]], 1
+; CHECK-NEXT:    [[TMP4:%.*]] = add i32 [[RHS_START:%.*]], [[TMP3]]
 ; CHECK-NEXT:    [[CMP1_EXIT:%.*]] = icmp sgt i32 [[TMP2]], [[TMP4]]
 ; CHECK-NEXT:    [[RESULT:%.*]] = and i1 [[CMP0_EXIT]], [[CMP1_EXIT]]
 ; CHECK-NEXT:    ret i1 [[RESULT]]
@@ -462,17 +458,18 @@ define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
-; CHECK-NEXT:    [[PTR:%.*]] = phi ptr [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[PTR_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[CMP:%.*]] = icmp ne ptr [[PTR]], [[END:%.*]]
-; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT:    br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
 ; CHECK:       body:
-; CHECK-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 1
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
 ; CHECK-NEXT:    br label [[LOOP]]
 ; CHECK:       exit:
+; CHECK-NEXT:    [[START2:%.*]] = ptrtoaddr ptr [[START:%.*]] to i64
+; CHECK-NEXT:    [[END1:%.*]] = ptrtoaddr ptr [[END:%.*]] to i64
+; CHECK-NEXT:    [[LIMIT_FR:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT:    [[TMP0:%.*]] = zext i32 [[LIMIT_FR]] to i64
+; CHECK-NEXT:    [[TMP1:%.*]] = sub i64 [[END1]], [[START2]]
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[START]], i64 [[UMIN]]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne ptr [[SCEVGEP]], [[END]]
 ; CHECK-NEXT:    ret i1 [[CMP]]
 ;
 entry:
diff --git a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
index f91972263d862..86f128debdb31 100644
--- a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
+++ b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
@@ -13,35 +13,15 @@ define i1 @rewrite_loop_exit_compare(i32 %length, i64 %limit) vscale_range(2, 10
 ; CHECK-NEXT:    [[CONTINUE3:%.*]] = and i1 [[NONZERO1]], [[INRANGE2]]
 ; CHECK-NEXT:    br i1 [[CONTINUE3]], label %[[BODY_PREHEADER:.*]], label %[[EXIT:.*]]
 ; CHECK:       [[BODY_PREHEADER]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
 ; CHECK-NEXT:    [[TMP0:%.*]] = add i32 [[LENGTH]], -1
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
 ; CHECK-NEXT:    [[TMP2:%.*]] = zext i32 [[TMP0]] to i64
 ; CHECK-NEXT:    [[UMIN:%.*]] = tail call i64 @llvm.umin.i64(i64 [[TMP1]], i64 [[TMP2]])
-; CHECK-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[UMIN]], 1
-; CHECK-NEXT:    [[TMP5:%.*]] = tail call <vscale x 16 x i32> @llvm.stepvector.nxv16i32()
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[LENGTH]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP6:%.*]] = sub nsw <vscale x 16 x i32> [[BROADCAST_SPLAT]], [[TMP5]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 16 x i32> [ [[TMP6]], %[[BODY_PREHEADER]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[AVL:%.*]] = phi i64 [ [[TMP4]], %[[BODY_PREHEADER]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP7:%.*]] = tail call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 16, i1 true)
-; CHECK-NEXT:    [[TMP8:%.*]] = zext i32 [[TMP7]] to i64
-; CHECK-NEXT:    [[TMP9:%.*]] = sub nsw i32 0, [[TMP7]]
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT9:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[TMP9]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT10:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT9]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-NEXT:    [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP8]]
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nsw <vscale x 16 x i32> [[VEC_IND]], [[BROADCAST_SPLAT10]]
-; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
-; CHECK-NEXT:    br i1 [[TMP10]], label %[[EXIT_LOOPEXIT:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK:       [[EXIT_LOOPEXIT]]:
-; CHECK-NEXT:    [[TMP11:%.*]] = add nsw i64 [[TMP8]], -1
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <vscale x 16 x i32> [[VEC_IND]], i64 [[TMP11]]
-; CHECK-NEXT:    [[TMP13:%.*]] = icmp ne i32 [[TMP12]], 1
+; CHECK-NEXT:    [[TMP3:%.*]] = trunc nuw i64 [[UMIN]] to i32
+; CHECK-NEXT:    [[NONZERO_EXIT:%.*]] = icmp ne i32 [[TMP0]], [[TMP3]]
 ; CHECK-NEXT:    br label %[[EXIT]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[TMP13]], %[[EXIT_LOOPEXIT]] ]
+; CHECK-NEXT:    [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[NONZERO_EXIT]], %[[BODY_PREHEADER]] ]
 ; CHECK-NEXT:    ret i1 [[NONZERO_LCSSA]]
 ;
 entry:
@@ -63,8 +43,3 @@ body:
 exit:
   ret i1 %nonzero
 }
-;.
-; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
-; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
-; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
-;.

>From b8b261472315495836581cc87ce168210f1dfe8f Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 18:43:27 +0800
Subject: [PATCH 3/3] [IndVars] Preserve potentially infinite subloops

Only rebuild loop-exit comparisons when LoopDeletion's progress checks
allow the loop and its subloops to be removed. This avoids making a
potentially infinite subloop unreachable.

Assisted-by: TRAE CLI (GPT-5)
---
 llvm/lib/Transforms/Utils/LoopUtils.cpp       |  25 +++-
 .../IndVarSimplify/rewrite-loop-exit-value.ll | 138 ++++++++++++++++--
 2 files changed, 147 insertions(+), 16 deletions(-)

diff --git a/llvm/lib/Transforms/Utils/LoopUtils.cpp b/llvm/lib/Transforms/Utils/LoopUtils.cpp
index 5c1a6fbaebbdd..1218098369522 100644
--- a/llvm/lib/Transforms/Utils/LoopUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopUtils.cpp
@@ -19,11 +19,13 @@
 #include "llvm/ADT/SmallVector.h"
 #include "llvm/Analysis/AliasAnalysis.h"
 #include "llvm/Analysis/BasicAliasAnalysis.h"
+#include "llvm/Analysis/CFG.h"
 #include "llvm/Analysis/DomTreeUpdater.h"
 #include "llvm/Analysis/GlobalsModRef.h"
 #include "llvm/Analysis/InstSimplifyFolder.h"
 #include "llvm/Analysis/LoopAccessAnalysis.h"
 #include "llvm/Analysis/LoopInfo.h"
+#include "llvm/Analysis/LoopIterator.h"
 #include "llvm/Analysis/LoopPass.h"
 #include "llvm/Analysis/MemorySSA.h"
 #include "llvm/Analysis/MemorySSAUpdater.h"
@@ -1815,7 +1817,8 @@ struct RewritePhi {
 // Check whether it is possible to delete the loop after rewriting exit
 // value. If it is possible, ignore ReplaceExitValue and do rewriting
 // aggressively.
-static bool canLoopBeDeleted(Loop *L, SmallVector<RewritePhi, 8> &RewritePhiSet) {
+static bool canLoopBeDeleted(Loop *L, SmallVector<RewritePhi, 8> &RewritePhiSet,
+                             ScalarEvolution *SE, LoopInfo *LI) {
   BasicBlock *Preheader = L->getLoopPreheader();
   // If there is no preheader, the loop will not be deleted.
   if (!Preheader)
@@ -1863,6 +1866,24 @@ static bool canLoopBeDeleted(Loop *L, SmallVector<RewritePhi, 8> &RewritePhiSet)
         }))
       return false;
 
+  if (L->getHeader()->getParent()->mustProgress())
+    return true;
+
+  LoopBlocksRPO RPOT(L);
+  RPOT.perform(LI);
+  if (containsIrreducibleCFG<const BasicBlock *>(RPOT, *LI))
+    return false;
+
+  SmallVector<Loop *, 8> WorkList;
+  WorkList.push_back(L);
+  while (!WorkList.empty()) {
+    Loop *Current = WorkList.pop_back_val();
+    if (hasMustProgress(Current))
+      continue;
+    if (isa<SCEVCouldNotCompute>(SE->getConstantMaxBackedgeTakenCount(Current)))
+      return false;
+    WorkList.append(Current->begin(), Current->end());
+  }
   return true;
 }
 
@@ -2060,7 +2081,7 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
   // calculate the cost of other SCEV's after expanding SCEV 'A', thus
   // potentially giving cost bonus to those other SCEV's?
 
-  bool LoopCanBeDel = canLoopBeDeleted(L, RewritePhiSet);
+  bool LoopCanBeDel = canLoopBeDeleted(L, RewritePhiSet, SE, LI);
   int NumReplaced = 0;
 
   // Transformation.
diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index d9677d69e48d5..afa4a34b28f9d 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -4,6 +4,7 @@
 ;; Test that loop's exit value is rewritten to its initial
 ;; value from loop preheader
 define i32 @test1(ptr %var) {
+;
 ; CHECK-LABEL: @test1(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[COND:%.*]] = icmp eq ptr [[VAR:%.*]], null
@@ -34,6 +35,7 @@ exit:
 ;; Test that we can not rewrite loop exit value if it's not
 ;; a phi node (%indvar is an add instruction in this test).
 define i32 @test2(ptr %var) {
+;
 ; CHECK-LABEL: @test2(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[COND:%.*]] = icmp eq ptr [[VAR:%.*]], null
@@ -61,6 +63,7 @@ exit:
 ;; Test that we can not rewrite loop exit value if the condition
 ;; is not in loop header.
 define i32 @test3(ptr %var) {
+;
 ; CHECK-LABEL: @test3(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[COND1:%.*]] = icmp eq ptr [[VAR:%.*]], null
@@ -97,6 +100,7 @@ exit:
 
 ; Multiple exits dominating latch
 define i32 @test4(i1 %cond1, i1 %cond2) {
+;
 ; CHECK-LABEL: @test4(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[HEADER:%.*]]
@@ -124,6 +128,7 @@ exit:
 
 ; A conditionally executed exit.
 define i32 @test5(ptr %addr, i1 %cond2) {
+;
 ; CHECK-LABEL: @test5(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[HEADER:%.*]]
@@ -159,6 +164,7 @@ exit:
 }
 
 define i16 @pr57336(i16 %end, i16 %m) mustprogress {
+;
 ; CHECK-LABEL: @pr57336(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[FOR_BODY:%.*]]
@@ -198,6 +204,7 @@ crit_edge:
 }
 
 define i32 @vscale_slt_with_vp_umin(ptr nocapture %A, i32 %n) mustprogress vscale_range(2,1024) {
+;
 ; CHECK-LABEL: @vscale_slt_with_vp_umin(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[VSCALE:%.*]] = call i32 @llvm.vscale.i32()
@@ -218,12 +225,12 @@ define i32 @vscale_slt_with_vp_umin(ptr nocapture %A, i32 %n) mustprogress vscal
 ; CHECK-NEXT:    br i1 [[CMP]], label [[FOR_BODY]], label [[FOR_END:%.*]]
 ; CHECK:       for.end:
 ; CHECK-NEXT:    [[TMP0:%.*]] = add nsw i32 [[N]], -1
-; CHECK-NEXT:    [[TMP5:%.*]] = call range(i32 2, 33) i32 @llvm.cttz.i32(i32 [[VF]], i1 true)
-; CHECK-NEXT:    [[TMP1:%.*]] = lshr i32 [[TMP0]], [[TMP5]]
-; CHECK-NEXT:    [[TMP2:%.*]] = mul i32 [[TMP1]], [[VSCALE]]
-; CHECK-NEXT:    [[TMP3:%.*]] = shl i32 [[TMP2]], 2
-; CHECK-NEXT:    [[TMP4:%.*]] = sub i32 [[N]], [[TMP3]]
-; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[VF]], i32 [[TMP4]])
+; CHECK-NEXT:    [[TMP1:%.*]] = call range(i32 2, 33) i32 @llvm.cttz.i32(i32 [[VF]], i1 true)
+; CHECK-NEXT:    [[TMP2:%.*]] = lshr i32 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = mul i32 [[TMP2]], [[VSCALE]]
+; CHECK-NEXT:    [[TMP4:%.*]] = shl i32 [[TMP3]], 2
+; CHECK-NEXT:    [[TMP5:%.*]] = sub i32 [[N]], [[TMP4]]
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[VF]], i32 [[TMP5]])
 ; CHECK-NEXT:    ret i32 [[UMIN]]
 ;
 entry:
@@ -251,6 +258,7 @@ for.end:
 }
 
 define i32 @vscale_slt_with_vp_umin2(ptr nocapture %A, i32 %n) mustprogress vscale_range(2,1024) {
+;
 ; CHECK-LABEL: @vscale_slt_with_vp_umin2(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    [[VSCALE:%.*]] = call i32 @llvm.vscale.i32()
@@ -271,12 +279,12 @@ define i32 @vscale_slt_with_vp_umin2(ptr nocapture %A, i32 %n) mustprogress vsca
 ; CHECK-NEXT:    br i1 [[CMP]], label [[FOR_BODY]], label [[FOR_END:%.*]]
 ; CHECK:       for.end:
 ; CHECK-NEXT:    [[TMP0:%.*]] = add i32 [[N]], -1
-; CHECK-NEXT:    [[TMP5:%.*]] = call range(i32 2, 33) i32 @llvm.cttz.i32(i32 [[VF]], i1 true)
-; CHECK-NEXT:    [[TMP1:%.*]] = lshr i32 [[TMP0]], [[TMP5]]
-; CHECK-NEXT:    [[TMP2:%.*]] = mul i32 [[TMP1]], [[VSCALE]]
-; CHECK-NEXT:    [[TMP3:%.*]] = shl i32 [[TMP2]], 2
-; CHECK-NEXT:    [[TMP4:%.*]] = sub i32 [[N]], [[TMP3]]
-; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[VF]], i32 [[TMP4]])
+; CHECK-NEXT:    [[TMP1:%.*]] = call range(i32 2, 33) i32 @llvm.cttz.i32(i32 [[VF]], i1 true)
+; CHECK-NEXT:    [[TMP2:%.*]] = lshr i32 [[TMP0]], [[TMP1]]
+; CHECK-NEXT:    [[TMP3:%.*]] = mul i32 [[TMP2]], [[VSCALE]]
+; CHECK-NEXT:    [[TMP4:%.*]] = shl i32 [[TMP3]], 2
+; CHECK-NEXT:    [[TMP5:%.*]] = sub i32 [[N]], [[TMP4]]
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[VF]], i32 [[TMP5]])
 ; CHECK-NEXT:    ret i32 [[UMIN]]
 ;
 entry:
@@ -305,6 +313,7 @@ for.end:
 
 ; Rewrite a comparison by expanding its operands at the loop exit.
 define i1 @rewrite_computable_icmp(i32 %start, i32 %limit) {
+;
 ; CHECK-LABEL: @rewrite_computable_icmp(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -342,6 +351,7 @@ exit:
 
 ; Rewrite multiple comparisons when doing so makes all live-outs invariant.
 define i1 @rewrite_multiple_icmps(i32 %start, i32 %rhs.start, i32 %limit) {
+;
 ; CHECK-LABEL: @rewrite_multiple_icmps(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -387,6 +397,7 @@ exit:
 
 ; Do not rewrite a comparison if SCEV cannot compute an operand's exit value.
 define i1 @do_not_rewrite_uncomputable_icmp(i1 %c, i32 %start) {
+;
 ; CHECK-LABEL: @do_not_rewrite_uncomputable_icmp(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -415,6 +426,7 @@ exit:
 
 ; Do not rebuild a comparison unless doing so makes the loop deletable.
 define i1 @do_not_rewrite_icmp_in_live_loop(i32 %start, i32 %limit) {
+;
 ; CHECK-LABEL: @do_not_rewrite_icmp_in_live_loop(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -452,8 +464,103 @@ exit:
   ret i1 %cmp
 }
 
+; Do not rewrite a comparison if the loop contains a potentially infinite
+; subloop. Rewriting it would make the subloop unreachable and change whether
+; the function terminates.
+define i1 @do_not_rewrite_icmp_with_infinite_subloop(i32 %start, i32 %limit, i1 %c) {
+; CHECK-LABEL: @do_not_rewrite_icmp_with_infinite_subloop(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[OUTER_HEADER:%.*]]
+; CHECK:       outer.header:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[OUTER_LATCH:%.*]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[OUTER_LATCH]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i32 [[IV]], 42
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT:    br i1 [[CONTINUE]], label [[INNER_PREHEADER:%.*]], label [[EXIT:%.*]]
+; CHECK:       inner.preheader:
+; CHECK-NEXT:    br label [[INNER:%.*]]
+; CHECK:       inner:
+; CHECK-NEXT:    br i1 [[C:%.*]], label [[INNER]], label [[OUTER_LATCH]]
+; CHECK:       outer.latch:
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[OUTER_HEADER]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %outer.latch ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %outer.latch ]
+  %cmp = icmp ne i32 %iv, 42
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp, i1 %inrange, i1 false
+  br i1 %continue, label %inner, label %exit
+
+inner:
+  br i1 %c, label %inner, label %outer.latch
+
+outer.latch:
+  %iv.next = add nsw i32 %iv, -1
+  %index.next = add nuw i32 %index, 1
+  br label %outer.header
+
+exit:
+  ret i1 %cmp
+}
+
+; A mustprogress outer loop allows the same rewrite even if a subloop has an
+; unknown trip count.
+define i1 @rewrite_icmp_with_mustprogress_outer_loop(i32 %start, i32 %limit, i1 %c) {
+;
+; CHECK-LABEL: @rewrite_icmp_with_mustprogress_outer_loop(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[OUTER_HEADER:%.*]]
+; CHECK:       outer.header:
+; CHECK-NEXT:    br i1 false, label [[INNER_PREHEADER:%.*]], label [[EXIT:%.*]]
+; CHECK:       inner.preheader:
+; CHECK-NEXT:    br label [[INNER:%.*]]
+; CHECK:       inner:
+; CHECK-NEXT:    br i1 [[C:%.*]], label [[INNER]], label [[OUTER_LATCH:%.*]]
+; CHECK:       outer.latch:
+; CHECK-NEXT:    br label [[OUTER_HEADER]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       exit:
+; CHECK-NEXT:    [[TMP0:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[START:%.*]], -42
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[TMP0]], i32 [[TMP1]])
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[START]], [[UMIN]]
+; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
+; CHECK-NEXT:    ret i1 [[CMP_EXIT]]
+;
+entry:
+  br label %outer.header
+
+outer.header:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %outer.latch ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %outer.latch ]
+  %cmp = icmp ne i32 %iv, 42
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp, i1 %inrange, i1 false
+  br i1 %continue, label %inner, label %exit
+
+inner:
+  br i1 %c, label %inner, label %outer.latch
+
+outer.latch:
+  %iv.next = add nsw i32 %iv, -1
+  %index.next = add nuw i32 %index, 1
+  br label %outer.header, !llvm.loop !0
+
+exit:
+  ret i1 %cmp
+}
+
 ; Rewrite a pointer comparison when its operand exit values are computable.
 define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
+;
 ; CHECK-LABEL: @rewrite_pointer_icmp(
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
@@ -469,8 +576,8 @@ define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
 ; CHECK-NEXT:    [[TMP1:%.*]] = sub i64 [[END1]], [[START2]]
 ; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 [[TMP1]])
 ; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[START]], i64 [[UMIN]]
-; CHECK-NEXT:    [[CMP:%.*]] = icmp ne ptr [[SCEVGEP]], [[END]]
-; CHECK-NEXT:    ret i1 [[CMP]]
+; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne ptr [[SCEVGEP]], [[END]]
+; CHECK-NEXT:    ret i1 [[CMP_EXIT]]
 ;
 entry:
   br label %loop
@@ -493,3 +600,6 @@ exit:
 }
 
 declare void @use.i1(i1)
+
+!0 = distinct !{!0, !1}
+!1 = !{!"llvm.loop.mustprogress"}



More information about the llvm-commits mailing list