[llvm] [LICM] Sink comparisons that enable loop deletion (PR #223951)

Pengcheng Wang via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 16 23:51:24 PDT 2026


https://github.com/wangpc-pp updated https://github.com/llvm/llvm-project/pull/223951

>From 31532d528e1c67110a63616b53edddab203b1737 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 12:58:40 +0800
Subject: [PATCH 1/2] [IndVars] Add tests for rewriting loop-exit comparisons

Add coverage for integer and pointer comparisons, multiple varying

operands, uncomputable exit values, and the RISC-V phase-ordering

issue.

Assisted-by: TRAE CLI (GPT-5)
---
 .../IndVarSimplify/rewrite-loop-exit-value.ll | 194 ++++++++++++++++++
 .../RISCV/rewrite-loop-exit-compare.ll        |  70 +++++++
 2 files changed, 264 insertions(+)
 create mode 100644 llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll

diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index fa47d06d859e9..09e8765610f69 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -302,3 +302,197 @@ for.body:
 for.end:
   ret i32 %VF.capped
 }
+
+; Rewrite a comparison by expanding its operands at the loop exit.
+define i1 @rewrite_computable_icmp(i32 %start, i32 %limit) {
+; CHECK-LABEL: @rewrite_computable_icmp(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[IV]], 42
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP_EXIT]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP_EXIT]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp = icmp ne i32 %iv, 42
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %iv.next = add nsw i32 %iv, -1
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %cmp
+}
+
+; Rewrite multiple comparisons when doing so makes all live-outs invariant.
+define i1 @rewrite_multiple_icmps(i32 %start, i32 %rhs.start, i32 %limit) {
+; CHECK-LABEL: @rewrite_multiple_icmps(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[TMP2:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[TMP4:%.*]] = phi i32 [ [[RHS_START:%.*]], [[ENTRY]] ], [ [[RHS_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP0_EXIT]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[TMP2]], -1
+; CHECK-NEXT:    [[RHS_NEXT]] = add i32 [[TMP4]], 2
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    [[CMP1_EXIT:%.*]] = icmp sgt i32 [[TMP2]], [[TMP4]]
+; CHECK-NEXT:    [[RESULT:%.*]] = and i1 [[CMP0_EXIT]], [[CMP1_EXIT]]
+; CHECK-NEXT:    ret i1 [[RESULT]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+  %rhs = phi i32 [ %rhs.start, %entry ], [ %rhs.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp0 = icmp ne i32 %iv, 42
+  %cmp1 = icmp sgt i32 %iv, %rhs
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp0, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %iv.next = add nsw i32 %iv, -1
+  %rhs.next = add i32 %rhs, 2
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  %result = and i1 %cmp0, %cmp1
+  ret i1 %result
+}
+
+; Do not rewrite a comparison if SCEV cannot compute an operand's exit value.
+define i1 @do_not_rewrite_uncomputable_icmp(i1 %c, i32 %start) {
+; CHECK-LABEL: @do_not_rewrite_uncomputable_icmp(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i32 [[IV]], 0
+; CHECK-NEXT:    [[SELECTED:%.*]] = select i1 [[CMP]], i1 [[C:%.*]], i1 false
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT:    br i1 [[SELECTED]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %loop ]
+  %cmp = icmp ne i32 %iv, 0
+  %selected = select i1 %cmp, i1 %c, i1 false
+  %iv.next = add i32 %iv, 1
+  br i1 %selected, label %loop, label %exit
+
+exit:
+  ret i1 %cmp
+}
+
+; Do not rebuild a comparison unless doing so makes the loop deletable.
+define i1 @do_not_rewrite_icmp_in_live_loop(i32 %start, i32 %limit) {
+; CHECK-LABEL: @do_not_rewrite_icmp_in_live_loop(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne i32 [[IV]], 42
+; CHECK-NEXT:    call void @use.i1(i1 [[CMP]])
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    br i1 [[INRANGE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp = icmp ne i32 %iv, 42
+  call void @use.i1(i1 %cmp)
+  %inrange = icmp ult i32 %index, %limit
+  br i1 %inrange, label %body, label %exit
+
+body:
+  %iv.next = add nsw i32 %iv, -1
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %cmp
+}
+
+; Rewrite a pointer comparison when its operand exit values are computable.
+define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
+; CHECK-LABEL: @rewrite_pointer_icmp(
+; CHECK-NEXT:  entry:
+; CHECK-NEXT:    br label [[LOOP:%.*]]
+; CHECK:       loop:
+; CHECK-NEXT:    [[PTR:%.*]] = phi ptr [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[PTR_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne ptr [[PTR]], [[END:%.*]]
+; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK:       body:
+; CHECK-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 1
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT:    br label [[LOOP]]
+; CHECK:       exit:
+; CHECK-NEXT:    ret i1 [[CMP]]
+;
+entry:
+  br label %loop
+
+loop:
+  %ptr = phi ptr [ %start, %entry ], [ %ptr.next, %body ]
+  %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+  %cmp = icmp ne ptr %ptr, %end
+  %inrange = icmp ult i32 %index, %limit
+  %continue = select i1 %cmp, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %ptr.next = getelementptr i8, ptr %ptr, i64 1
+  %index.next = add nuw i32 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %cmp
+}
+
+declare void @use.i1(i1)
diff --git a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
new file mode 100644
index 0000000000000..f91972263d862
--- /dev/null
+++ b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
@@ -0,0 +1,70 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -mtriple=riscv64 -mattr=+v -passes='default<O3>' -S %s | FileCheck %s
+
+; Rewriting the comparison from its operands' exit values allows the loop to be
+; deleted instead of vectorizing the induction and extracting its final result.
+define i1 @rewrite_loop_exit_compare(i32 %length, i64 %limit) vscale_range(2, 1024) {
+; CHECK-LABEL: define i1 @rewrite_loop_exit_compare(
+; CHECK-SAME: i32 [[LENGTH:%.*]], i64 [[LIMIT:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[LIMIT_FR:%.*]] = freeze i64 [[LIMIT]]
+; CHECK-NEXT:    [[NONZERO1:%.*]] = icmp ne i32 [[LENGTH]], 0
+; CHECK-NEXT:    [[INRANGE2:%.*]] = icmp ne i64 [[LIMIT_FR]], 0
+; CHECK-NEXT:    [[CONTINUE3:%.*]] = and i1 [[NONZERO1]], [[INRANGE2]]
+; CHECK-NEXT:    br i1 [[CONTINUE3]], label %[[BODY_PREHEADER:.*]], label %[[EXIT:.*]]
+; CHECK:       [[BODY_PREHEADER]]:
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
+; CHECK-NEXT:    [[TMP0:%.*]] = add i32 [[LENGTH]], -1
+; CHECK-NEXT:    [[TMP2:%.*]] = zext i32 [[TMP0]] to i64
+; CHECK-NEXT:    [[UMIN:%.*]] = tail call i64 @llvm.umin.i64(i64 [[TMP1]], i64 [[TMP2]])
+; CHECK-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[UMIN]], 1
+; CHECK-NEXT:    [[TMP5:%.*]] = tail call <vscale x 16 x i32> @llvm.stepvector.nxv16i32()
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[LENGTH]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-NEXT:    [[TMP6:%.*]] = sub nsw <vscale x 16 x i32> [[BROADCAST_SPLAT]], [[TMP5]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 16 x i32> [ [[TMP6]], %[[BODY_PREHEADER]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[AVL:%.*]] = phi i64 [ [[TMP4]], %[[BODY_PREHEADER]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP7:%.*]] = tail call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 16, i1 true)
+; CHECK-NEXT:    [[TMP8:%.*]] = zext i32 [[TMP7]] to i64
+; CHECK-NEXT:    [[TMP9:%.*]] = sub nsw i32 0, [[TMP7]]
+; CHECK-NEXT:    [[BROADCAST_SPLATINSERT9:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[TMP9]], i64 0
+; CHECK-NEXT:    [[BROADCAST_SPLAT10:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT9]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-NEXT:    [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP8]]
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nsw <vscale x 16 x i32> [[VEC_IND]], [[BROADCAST_SPLAT10]]
+; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
+; CHECK-NEXT:    br i1 [[TMP10]], label %[[EXIT_LOOPEXIT:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[EXIT_LOOPEXIT]]:
+; CHECK-NEXT:    [[TMP11:%.*]] = add nsw i64 [[TMP8]], -1
+; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <vscale x 16 x i32> [[VEC_IND]], i64 [[TMP11]]
+; CHECK-NEXT:    [[TMP13:%.*]] = icmp ne i32 [[TMP12]], 1
+; CHECK-NEXT:    br label %[[EXIT]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[TMP13]], %[[EXIT_LOOPEXIT]] ]
+; CHECK-NEXT:    ret i1 [[NONZERO_LCSSA]]
+;
+entry:
+  br label %loop
+
+loop:
+  %length.addr = phi i32 [ %length, %entry ], [ %dec, %body ]
+  %index = phi i64 [ 0, %entry ], [ %index.next, %body ]
+  %nonzero = icmp ne i32 %length.addr, 0
+  %inrange = icmp ult i64 %index, %limit
+  %continue = select i1 %nonzero, i1 %inrange, i1 false
+  br i1 %continue, label %body, label %exit
+
+body:
+  %dec = add nsw i32 %length.addr, -1
+  %index.next = add nuw i64 %index, 1
+  br label %loop
+
+exit:
+  ret i1 %nonzero
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+;.

>From 276cb73e2b17e56fb1d11db1d16f4eaef7895992 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 13:00:15 +0800
Subject: [PATCH 2/2] [IndVars] Rewrite loop-exit comparisons from operands

When SCEV cannot model a comparison, clone it in the loop preheader

using the exit values of its varying operands. This lets exit-value

rewriting remove otherwise dead loops before vectorization.

Assisted-by: TRAE CLI (GPT-5)
---
 llvm/lib/Transforms/Utils/LoopUtils.cpp       | 87 +++++++++++++++----
 .../IndVarSimplify/rewrite-loop-exit-value.ll | 49 +++++------
 .../RISCV/rewrite-loop-exit-compare.ll        | 33 +------
 3 files changed, 97 insertions(+), 72 deletions(-)

diff --git a/llvm/lib/Transforms/Utils/LoopUtils.cpp b/llvm/lib/Transforms/Utils/LoopUtils.cpp
index a2e544801b9c4..5c1a6fbaebbdd 100644
--- a/llvm/lib/Transforms/Utils/LoopUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopUtils.cpp
@@ -1790,6 +1790,10 @@ static bool hasHardUserWithinLoop(const Loop *L, const Instruction *I) {
   return false;
 }
 
+// Exit values of an instruction's loop-variant operands, indexed by operand
+// number.
+using OperandExitValueList = SmallVector<std::pair<unsigned, SCEVUse>, 2>;
+
 // Collect information about PHI nodes which can be transformed in
 // rewriteLoopExitValues.
 struct RewritePhi {
@@ -1798,11 +1802,14 @@ struct RewritePhi {
   SCEVUse ExpansionSCEV;     // The SCEV of the incoming value we are rewriting.
   Instruction *ExpansionPoint; // Where we'd like to expand that SCEV?
   bool HighCost;               // Is this expansion a high-cost?
+  // If SCEV cannot model the incoming instruction, use these values to clone
+  // the instruction with its loop-variant operands rewritten.
+  OperandExitValueList OperandExitValues;
 
   RewritePhi(PHINode *P, unsigned I, SCEVUse Val, Instruction *ExpansionPt,
-             bool H)
+             bool H, OperandExitValueList OperandExitValues)
       : PN(P), Ith(I), ExpansionSCEV(Val), ExpansionPoint(ExpansionPt),
-        HighCost(H) {}
+        HighCost(H), OperandExitValues(std::move(OperandExitValues)) {}
 };
 
 // Check whether it is possible to delete the loop after rewriting exit
@@ -1873,6 +1880,29 @@ static bool checkIsIndPhi(PHINode *Phi, Loop *L, ScalarEvolution *SE,
   return InductionDescriptor::isInductionPHI(Phi, L, SE, ID);
 }
 
+static bool collectOperandExitValues(Instruction *Inst, Loop *L,
+                                     BasicBlock *Preheader, ScalarEvolution *SE,
+                                     SCEVExpander &Rewriter,
+                                     OperandExitValueList &ExitValues) {
+  if (!Preheader || !isa<ICmpInst>(Inst))
+    return false;
+
+  for (const auto &[Idx, Op] : enumerate(Inst->operands())) {
+    // Operands defined outside the loop already dominate the preheader.
+    if (L->isLoopInvariant(Op))
+      continue;
+    if (!SE->isSCEVable(Op->getType()))
+      return false;
+    SCEVUse ExitValue = SE->getSCEVAtScope(Op, L->getParentLoop());
+    if (isa<SCEVCouldNotCompute>(ExitValue) ||
+        !SE->isLoopInvariant(ExitValue, L) ||
+        !Rewriter.isSafeToExpandAt(ExitValue, Preheader->getTerminator()))
+      return false;
+    ExitValues.emplace_back(Idx, ExitValue);
+  }
+  return !ExitValues.empty();
+}
+
 int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
                                 ScalarEvolution *SE,
                                 const TargetTransformInfo *TTI,
@@ -1971,6 +2001,7 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
         // expression reuse by the SCEVExpander), but resort to per-exit
         // evaluation if that fails.
         SCEVUse ExitValue = SE->getSCEVAtScope(Inst, L->getParentLoop());
+        OperandExitValueList OperandExitValues;
         if (isa<SCEVCouldNotCompute>(ExitValue) ||
             !SE->isLoopInvariant(ExitValue, L) ||
             !Rewriter.isSafeToExpand(ExitValue)) {
@@ -1979,14 +2010,15 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
           // most SCEV expressions and other recurrence types (e.g. shift
           // recurrences).  Is there existing code we can reuse?
           const SCEV *ExitCount = SE->getExitCount(L, PN->getIncomingBlock(i));
-          if (isa<SCEVCouldNotCompute>(ExitCount))
-            continue;
-          if (auto *AddRec = dyn_cast<SCEVAddRecExpr>(SE->getSCEV(Inst)))
-            if (AddRec->getLoop() == L)
-              ExitValue = AddRec->evaluateAtIteration(ExitCount, *SE);
-          if (isa<SCEVCouldNotCompute>(ExitValue) ||
-              !SE->isLoopInvariant(ExitValue, L) ||
-              !Rewriter.isSafeToExpand(ExitValue))
+          if (!isa<SCEVCouldNotCompute>(ExitCount))
+            if (auto *AddRec = dyn_cast<SCEVAddRecExpr>(SE->getSCEV(Inst)))
+              if (AddRec->getLoop() == L)
+                ExitValue = AddRec->evaluateAtIteration(ExitCount, *SE);
+          if ((isa<SCEVCouldNotCompute>(ExitValue) ||
+               !SE->isLoopInvariant(ExitValue, L) ||
+               !Rewriter.isSafeToExpand(ExitValue)) &&
+              !collectOperandExitValues(Inst, L, L->getLoopPreheader(), SE,
+                                        Rewriter, OperandExitValues))
             continue;
         }
 
@@ -2000,8 +2032,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
           continue;
 
         // Check if expansions of this SCEV would count as being high cost.
-        bool HighCost = Rewriter.isHighCostExpansion(
-            ExitValue.getPointer(), L, SCEVCheapExpansionBudget, TTI, Inst);
+        bool HighCost =
+            OperandExitValues.empty() &&
+            Rewriter.isHighCostExpansion(ExitValue.getPointer(), L,
+                                         SCEVCheapExpansionBudget, TTI, Inst);
 
         // Note that we must not perform expansions until after
         // we query *all* the costs, because if we perform temporary expansion
@@ -2013,7 +2047,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
         Instruction *InsertPt =
           (isa<PHINode>(Inst) || isa<LandingPadInst>(Inst)) ?
           &*Inst->getParent()->getFirstInsertionPt() : Inst;
-        RewritePhiSet.emplace_back(PN, i, ExitValue, InsertPt, HighCost);
+        if (!OperandExitValues.empty())
+          InsertPt = L->getLoopPreheader()->getTerminator();
+        RewritePhiSet.emplace_back(PN, i, ExitValue, InsertPt, HighCost,
+                                   std::move(OperandExitValues));
       }
     }
   }
@@ -2030,6 +2067,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
   for (const RewritePhi &Phi : RewritePhiSet) {
     PHINode *PN = Phi.PN;
 
+    // Only clone an instruction when doing so allows the loop to be deleted.
+    if (!LoopCanBeDel && !Phi.OperandExitValues.empty())
+      continue;
+
     // Only do the rewrite when the ExitValue can be expanded cheaply.
     // If LoopCanBeDel is true, rewrite exit value aggressively.
     if ((ReplaceExitValue == OnlyCheapRepl ||
@@ -2037,12 +2078,25 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
         !LoopCanBeDel && Phi.HighCost)
       continue;
 
-    Value *ExitVal = Rewriter.expandCodeFor(
-        Phi.ExpansionSCEV, Phi.PN->getType(), Phi.ExpansionPoint);
+    Instruction *Inst = cast<Instruction>(PN->getIncomingValue(Phi.Ith));
+    Value *ExitVal;
+    if (Phi.OperandExitValues.empty()) {
+      ExitVal = Rewriter.expandCodeFor(Phi.ExpansionSCEV, Phi.PN->getType(),
+                                       Phi.ExpansionPoint);
+    } else {
+      Instruction *Clone = Inst->clone();
+      Clone->setName(Inst->getName() + ".exit");
+      for (auto [Idx, ExitSCEV] : Phi.OperandExitValues)
+        Clone->setOperand(Idx, Rewriter.expandCodeFor(
+                                   ExitSCEV, Inst->getOperand(Idx)->getType(),
+                                   Phi.ExpansionPoint));
+      Clone->insertBefore(Phi.ExpansionPoint->getIterator());
+      ExitVal = Clone;
+    }
 
     LLVM_DEBUG(dbgs() << "rewriteLoopExitValues: AfterLoopVal = " << *ExitVal
                       << '\n'
-                      << "  LoopVal = " << *(Phi.ExpansionPoint) << "\n");
+                      << "  LoopVal = " << *Inst << "\n");
 
 #ifndef NDEBUG
     // If we reuse an instruction from a loop which is neither L nor one of
@@ -2055,7 +2109,6 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
 #endif
 
     NumReplaced++;
-    Instruction *Inst = cast<Instruction>(PN->getIncomingValue(Phi.Ith));
     PN->setIncomingValue(Phi.Ith, ExitVal);
     // It's necessary to tell ScalarEvolution about this explicitly so that
     // it can walk the def-use list and forget all SCEVs, as it may not be
diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index 09e8765610f69..d9677d69e48d5 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -309,17 +309,15 @@ define i1 @rewrite_computable_icmp(i32 %start, i32 %limit) {
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
-; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[IV]], 42
-; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP_EXIT]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT:    br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
 ; CHECK:       body:
-; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[IV]], -1
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
 ; CHECK-NEXT:    br label [[LOOP]]
 ; CHECK:       exit:
+; CHECK-NEXT:    [[TMP0:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[START:%.*]], -42
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[TMP0]], i32 [[TMP1]])
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[START]], [[UMIN]]
+; CHECK-NEXT:    [[CMP_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
 ; CHECK-NEXT:    ret i1 [[CMP_EXIT]]
 ;
 entry:
@@ -348,19 +346,17 @@ define i1 @rewrite_multiple_icmps(i32 %start, i32 %rhs.start, i32 %limit) {
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
-; CHECK-NEXT:    [[TMP2:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT:    [[TMP4:%.*]] = phi i32 [ [[RHS_START:%.*]], [[ENTRY]] ], [ [[RHS_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
-; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP0_EXIT]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT:    br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
 ; CHECK:       body:
-; CHECK-NEXT:    [[IV_NEXT]] = add nsw i32 [[TMP2]], -1
-; CHECK-NEXT:    [[RHS_NEXT]] = add i32 [[TMP4]], 2
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
 ; CHECK-NEXT:    br label [[LOOP]]
 ; CHECK:       exit:
+; CHECK-NEXT:    [[TMP0:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT:    [[TMP1:%.*]] = add i32 [[START:%.*]], -42
+; CHECK-NEXT:    [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[TMP0]], i32 [[TMP1]])
+; CHECK-NEXT:    [[TMP2:%.*]] = sub i32 [[START]], [[UMIN]]
+; CHECK-NEXT:    [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
+; CHECK-NEXT:    [[TMP3:%.*]] = shl i32 [[UMIN]], 1
+; CHECK-NEXT:    [[TMP4:%.*]] = add i32 [[RHS_START:%.*]], [[TMP3]]
 ; CHECK-NEXT:    [[CMP1_EXIT:%.*]] = icmp sgt i32 [[TMP2]], [[TMP4]]
 ; CHECK-NEXT:    [[RESULT:%.*]] = and i1 [[CMP0_EXIT]], [[CMP1_EXIT]]
 ; CHECK-NEXT:    ret i1 [[RESULT]]
@@ -462,17 +458,18 @@ define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
 ; CHECK-NEXT:  entry:
 ; CHECK-NEXT:    br label [[LOOP:%.*]]
 ; CHECK:       loop:
-; CHECK-NEXT:    [[PTR:%.*]] = phi ptr [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[PTR_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT:    [[CMP:%.*]] = icmp ne ptr [[PTR]], [[END:%.*]]
-; CHECK-NEXT:    [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT:    [[CONTINUE:%.*]] = select i1 [[CMP]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT:    br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT:    br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
 ; CHECK:       body:
-; CHECK-NEXT:    [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 1
-; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
 ; CHECK-NEXT:    br label [[LOOP]]
 ; CHECK:       exit:
+; CHECK-NEXT:    [[START2:%.*]] = ptrtoaddr ptr [[START:%.*]] to i64
+; CHECK-NEXT:    [[END1:%.*]] = ptrtoaddr ptr [[END:%.*]] to i64
+; CHECK-NEXT:    [[LIMIT_FR:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT:    [[TMP0:%.*]] = zext i32 [[LIMIT_FR]] to i64
+; CHECK-NEXT:    [[TMP1:%.*]] = sub i64 [[END1]], [[START2]]
+; CHECK-NEXT:    [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT:    [[SCEVGEP:%.*]] = getelementptr i8, ptr [[START]], i64 [[UMIN]]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp ne ptr [[SCEVGEP]], [[END]]
 ; CHECK-NEXT:    ret i1 [[CMP]]
 ;
 entry:
diff --git a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
index f91972263d862..86f128debdb31 100644
--- a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
+++ b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
@@ -13,35 +13,15 @@ define i1 @rewrite_loop_exit_compare(i32 %length, i64 %limit) vscale_range(2, 10
 ; CHECK-NEXT:    [[CONTINUE3:%.*]] = and i1 [[NONZERO1]], [[INRANGE2]]
 ; CHECK-NEXT:    br i1 [[CONTINUE3]], label %[[BODY_PREHEADER:.*]], label %[[EXIT:.*]]
 ; CHECK:       [[BODY_PREHEADER]]:
-; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
 ; CHECK-NEXT:    [[TMP0:%.*]] = add i32 [[LENGTH]], -1
+; CHECK-NEXT:    [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
 ; CHECK-NEXT:    [[TMP2:%.*]] = zext i32 [[TMP0]] to i64
 ; CHECK-NEXT:    [[UMIN:%.*]] = tail call i64 @llvm.umin.i64(i64 [[TMP1]], i64 [[TMP2]])
-; CHECK-NEXT:    [[TMP4:%.*]] = add nuw nsw i64 [[UMIN]], 1
-; CHECK-NEXT:    [[TMP5:%.*]] = tail call <vscale x 16 x i32> @llvm.stepvector.nxv16i32()
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[LENGTH]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP6:%.*]] = sub nsw <vscale x 16 x i32> [[BROADCAST_SPLAT]], [[TMP5]]
-; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
-; CHECK:       [[VECTOR_BODY]]:
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <vscale x 16 x i32> [ [[TMP6]], %[[BODY_PREHEADER]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[AVL:%.*]] = phi i64 [ [[TMP4]], %[[BODY_PREHEADER]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP7:%.*]] = tail call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 16, i1 true)
-; CHECK-NEXT:    [[TMP8:%.*]] = zext i32 [[TMP7]] to i64
-; CHECK-NEXT:    [[TMP9:%.*]] = sub nsw i32 0, [[TMP7]]
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT9:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[TMP9]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT10:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT9]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-NEXT:    [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP8]]
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nsw <vscale x 16 x i32> [[VEC_IND]], [[BROADCAST_SPLAT10]]
-; CHECK-NEXT:    [[TMP10:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
-; CHECK-NEXT:    br i1 [[TMP10]], label %[[EXIT_LOOPEXIT:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK:       [[EXIT_LOOPEXIT]]:
-; CHECK-NEXT:    [[TMP11:%.*]] = add nsw i64 [[TMP8]], -1
-; CHECK-NEXT:    [[TMP12:%.*]] = extractelement <vscale x 16 x i32> [[VEC_IND]], i64 [[TMP11]]
-; CHECK-NEXT:    [[TMP13:%.*]] = icmp ne i32 [[TMP12]], 1
+; CHECK-NEXT:    [[TMP3:%.*]] = trunc nuw i64 [[UMIN]] to i32
+; CHECK-NEXT:    [[NONZERO_EXIT:%.*]] = icmp ne i32 [[TMP0]], [[TMP3]]
 ; CHECK-NEXT:    br label %[[EXIT]]
 ; CHECK:       [[EXIT]]:
-; CHECK-NEXT:    [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[TMP13]], %[[EXIT_LOOPEXIT]] ]
+; CHECK-NEXT:    [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[NONZERO_EXIT]], %[[BODY_PREHEADER]] ]
 ; CHECK-NEXT:    ret i1 [[NONZERO_LCSSA]]
 ;
 entry:
@@ -63,8 +43,3 @@ body:
 exit:
   ret i1 %nonzero
 }
-;.
-; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
-; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
-; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
-;.



More information about the llvm-commits mailing list