[llvm] [LICM] Sink comparisons that enable loop deletion (PR #223951)
Pengcheng Wang via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 16 23:51:24 PDT 2026
https://github.com/wangpc-pp updated https://github.com/llvm/llvm-project/pull/223951
>From 31532d528e1c67110a63616b53edddab203b1737 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 12:58:40 +0800
Subject: [PATCH 1/2] [IndVars] Add tests for rewriting loop-exit comparisons
Add coverage for integer and pointer comparisons, multiple varying
operands, uncomputable exit values, and the RISC-V phase-ordering
issue.
Assisted-by: TRAE CLI (GPT-5)
---
.../IndVarSimplify/rewrite-loop-exit-value.ll | 194 ++++++++++++++++++
.../RISCV/rewrite-loop-exit-compare.ll | 70 +++++++
2 files changed, 264 insertions(+)
create mode 100644 llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index fa47d06d859e9..09e8765610f69 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -302,3 +302,197 @@ for.body:
for.end:
ret i32 %VF.capped
}
+
+; Rewrite a comparison by expanding its operands at the loop exit.
+define i1 @rewrite_computable_icmp(i32 %start, i32 %limit) {
+; CHECK-LABEL: @rewrite_computable_icmp(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[LOOP:%.*]]
+; CHECK: loop:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT: [[CMP_EXIT:%.*]] = icmp ne i32 [[IV]], 42
+; CHECK-NEXT: [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT: [[CONTINUE:%.*]] = select i1 [[CMP_EXIT]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT: br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK: body:
+; CHECK-NEXT: [[IV_NEXT]] = add nsw i32 [[IV]], -1
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT: br label [[LOOP]]
+; CHECK: exit:
+; CHECK-NEXT: ret i1 [[CMP_EXIT]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+ %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+ %cmp = icmp ne i32 %iv, 42
+ %inrange = icmp ult i32 %index, %limit
+ %continue = select i1 %cmp, i1 %inrange, i1 false
+ br i1 %continue, label %body, label %exit
+
+body:
+ %iv.next = add nsw i32 %iv, -1
+ %index.next = add nuw i32 %index, 1
+ br label %loop
+
+exit:
+ ret i1 %cmp
+}
+
+; Rewrite multiple comparisons when doing so makes all live-outs invariant.
+define i1 @rewrite_multiple_icmps(i32 %start, i32 %rhs.start, i32 %limit) {
+; CHECK-LABEL: @rewrite_multiple_icmps(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[LOOP:%.*]]
+; CHECK: loop:
+; CHECK-NEXT: [[TMP2:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT: [[TMP4:%.*]] = phi i32 [ [[RHS_START:%.*]], [[ENTRY]] ], [ [[RHS_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT: [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
+; CHECK-NEXT: [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT: [[CONTINUE:%.*]] = select i1 [[CMP0_EXIT]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT: br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK: body:
+; CHECK-NEXT: [[IV_NEXT]] = add nsw i32 [[TMP2]], -1
+; CHECK-NEXT: [[RHS_NEXT]] = add i32 [[TMP4]], 2
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT: br label [[LOOP]]
+; CHECK: exit:
+; CHECK-NEXT: [[CMP1_EXIT:%.*]] = icmp sgt i32 [[TMP2]], [[TMP4]]
+; CHECK-NEXT: [[RESULT:%.*]] = and i1 [[CMP0_EXIT]], [[CMP1_EXIT]]
+; CHECK-NEXT: ret i1 [[RESULT]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+ %rhs = phi i32 [ %rhs.start, %entry ], [ %rhs.next, %body ]
+ %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+ %cmp0 = icmp ne i32 %iv, 42
+ %cmp1 = icmp sgt i32 %iv, %rhs
+ %inrange = icmp ult i32 %index, %limit
+ %continue = select i1 %cmp0, i1 %inrange, i1 false
+ br i1 %continue, label %body, label %exit
+
+body:
+ %iv.next = add nsw i32 %iv, -1
+ %rhs.next = add i32 %rhs, 2
+ %index.next = add nuw i32 %index, 1
+ br label %loop
+
+exit:
+ %result = and i1 %cmp0, %cmp1
+ ret i1 %result
+}
+
+; Do not rewrite a comparison if SCEV cannot compute an operand's exit value.
+define i1 @do_not_rewrite_uncomputable_icmp(i1 %c, i32 %start) {
+; CHECK-LABEL: @do_not_rewrite_uncomputable_icmp(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[LOOP:%.*]]
+; CHECK: loop:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[LOOP]] ]
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne i32 [[IV]], 0
+; CHECK-NEXT: [[SELECTED:%.*]] = select i1 [[CMP]], i1 [[C:%.*]], i1 false
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT: br i1 [[SELECTED]], label [[LOOP]], label [[EXIT:%.*]]
+; CHECK: exit:
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ %start, %entry ], [ %iv.next, %loop ]
+ %cmp = icmp ne i32 %iv, 0
+ %selected = select i1 %cmp, i1 %c, i1 false
+ %iv.next = add i32 %iv, 1
+ br i1 %selected, label %loop, label %exit
+
+exit:
+ ret i1 %cmp
+}
+
+; Do not rebuild a comparison unless doing so makes the loop deletable.
+define i1 @do_not_rewrite_icmp_in_live_loop(i32 %start, i32 %limit) {
+; CHECK-LABEL: @do_not_rewrite_icmp_in_live_loop(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[LOOP:%.*]]
+; CHECK: loop:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne i32 [[IV]], 42
+; CHECK-NEXT: call void @use.i1(i1 [[CMP]])
+; CHECK-NEXT: [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT: br i1 [[INRANGE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK: body:
+; CHECK-NEXT: [[IV_NEXT]] = add nsw i32 [[IV]], -1
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT: br label [[LOOP]]
+; CHECK: exit:
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ %start, %entry ], [ %iv.next, %body ]
+ %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+ %cmp = icmp ne i32 %iv, 42
+ call void @use.i1(i1 %cmp)
+ %inrange = icmp ult i32 %index, %limit
+ br i1 %inrange, label %body, label %exit
+
+body:
+ %iv.next = add nsw i32 %iv, -1
+ %index.next = add nuw i32 %index, 1
+ br label %loop
+
+exit:
+ ret i1 %cmp
+}
+
+; Rewrite a pointer comparison when its operand exit values are computable.
+define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
+; CHECK-LABEL: @rewrite_pointer_icmp(
+; CHECK-NEXT: entry:
+; CHECK-NEXT: br label [[LOOP:%.*]]
+; CHECK: loop:
+; CHECK-NEXT: [[PTR:%.*]] = phi ptr [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[PTR_NEXT:%.*]], [[BODY:%.*]] ]
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne ptr [[PTR]], [[END:%.*]]
+; CHECK-NEXT: [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
+; CHECK-NEXT: [[CONTINUE:%.*]] = select i1 [[CMP]], i1 [[INRANGE]], i1 false
+; CHECK-NEXT: br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK: body:
+; CHECK-NEXT: [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 1
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
+; CHECK-NEXT: br label [[LOOP]]
+; CHECK: exit:
+; CHECK-NEXT: ret i1 [[CMP]]
+;
+entry:
+ br label %loop
+
+loop:
+ %ptr = phi ptr [ %start, %entry ], [ %ptr.next, %body ]
+ %index = phi i32 [ 0, %entry ], [ %index.next, %body ]
+ %cmp = icmp ne ptr %ptr, %end
+ %inrange = icmp ult i32 %index, %limit
+ %continue = select i1 %cmp, i1 %inrange, i1 false
+ br i1 %continue, label %body, label %exit
+
+body:
+ %ptr.next = getelementptr i8, ptr %ptr, i64 1
+ %index.next = add nuw i32 %index, 1
+ br label %loop
+
+exit:
+ ret i1 %cmp
+}
+
+declare void @use.i1(i1)
diff --git a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
new file mode 100644
index 0000000000000..f91972263d862
--- /dev/null
+++ b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
@@ -0,0 +1,70 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -mtriple=riscv64 -mattr=+v -passes='default<O3>' -S %s | FileCheck %s
+
+; Rewriting the comparison from its operands' exit values allows the loop to be
+; deleted instead of vectorizing the induction and extracting its final result.
+define i1 @rewrite_loop_exit_compare(i32 %length, i64 %limit) vscale_range(2, 1024) {
+; CHECK-LABEL: define i1 @rewrite_loop_exit_compare(
+; CHECK-SAME: i32 [[LENGTH:%.*]], i64 [[LIMIT:%.*]]) local_unnamed_addr #[[ATTR0:[0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[LIMIT_FR:%.*]] = freeze i64 [[LIMIT]]
+; CHECK-NEXT: [[NONZERO1:%.*]] = icmp ne i32 [[LENGTH]], 0
+; CHECK-NEXT: [[INRANGE2:%.*]] = icmp ne i64 [[LIMIT_FR]], 0
+; CHECK-NEXT: [[CONTINUE3:%.*]] = and i1 [[NONZERO1]], [[INRANGE2]]
+; CHECK-NEXT: br i1 [[CONTINUE3]], label %[[BODY_PREHEADER:.*]], label %[[EXIT:.*]]
+; CHECK: [[BODY_PREHEADER]]:
+; CHECK-NEXT: [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
+; CHECK-NEXT: [[TMP0:%.*]] = add i32 [[LENGTH]], -1
+; CHECK-NEXT: [[TMP2:%.*]] = zext i32 [[TMP0]] to i64
+; CHECK-NEXT: [[UMIN:%.*]] = tail call i64 @llvm.umin.i64(i64 [[TMP1]], i64 [[TMP2]])
+; CHECK-NEXT: [[TMP4:%.*]] = add nuw nsw i64 [[UMIN]], 1
+; CHECK-NEXT: [[TMP5:%.*]] = tail call <vscale x 16 x i32> @llvm.stepvector.nxv16i32()
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[LENGTH]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP6:%.*]] = sub nsw <vscale x 16 x i32> [[BROADCAST_SPLAT]], [[TMP5]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <vscale x 16 x i32> [ [[TMP6]], %[[BODY_PREHEADER]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[AVL:%.*]] = phi i64 [ [[TMP4]], %[[BODY_PREHEADER]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP7:%.*]] = tail call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 16, i1 true)
+; CHECK-NEXT: [[TMP8:%.*]] = zext i32 [[TMP7]] to i64
+; CHECK-NEXT: [[TMP9:%.*]] = sub nsw i32 0, [[TMP7]]
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT9:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[TMP9]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT10:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT9]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
+; CHECK-NEXT: [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP8]]
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add nsw <vscale x 16 x i32> [[VEC_IND]], [[BROADCAST_SPLAT10]]
+; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
+; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT_LOOPEXIT:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[EXIT_LOOPEXIT]]:
+; CHECK-NEXT: [[TMP11:%.*]] = add nsw i64 [[TMP8]], -1
+; CHECK-NEXT: [[TMP12:%.*]] = extractelement <vscale x 16 x i32> [[VEC_IND]], i64 [[TMP11]]
+; CHECK-NEXT: [[TMP13:%.*]] = icmp ne i32 [[TMP12]], 1
+; CHECK-NEXT: br label %[[EXIT]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[TMP13]], %[[EXIT_LOOPEXIT]] ]
+; CHECK-NEXT: ret i1 [[NONZERO_LCSSA]]
+;
+entry:
+ br label %loop
+
+loop:
+ %length.addr = phi i32 [ %length, %entry ], [ %dec, %body ]
+ %index = phi i64 [ 0, %entry ], [ %index.next, %body ]
+ %nonzero = icmp ne i32 %length.addr, 0
+ %inrange = icmp ult i64 %index, %limit
+ %continue = select i1 %nonzero, i1 %inrange, i1 false
+ br i1 %continue, label %body, label %exit
+
+body:
+ %dec = add nsw i32 %length.addr, -1
+ %index.next = add nuw i64 %index, 1
+ br label %loop
+
+exit:
+ ret i1 %nonzero
+}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+;.
>From 276cb73e2b17e56fb1d11db1d16f4eaef7895992 Mon Sep 17 00:00:00 2001
From: Pengcheng Wang <wangpengcheng.pp at bytedance.com>
Date: Thu, 17 Sep 2026 13:00:15 +0800
Subject: [PATCH 2/2] [IndVars] Rewrite loop-exit comparisons from operands
When SCEV cannot model a comparison, clone it in the loop preheader
using the exit values of its varying operands. This lets exit-value
rewriting remove otherwise dead loops before vectorization.
Assisted-by: TRAE CLI (GPT-5)
---
llvm/lib/Transforms/Utils/LoopUtils.cpp | 87 +++++++++++++++----
.../IndVarSimplify/rewrite-loop-exit-value.ll | 49 +++++------
.../RISCV/rewrite-loop-exit-compare.ll | 33 +------
3 files changed, 97 insertions(+), 72 deletions(-)
diff --git a/llvm/lib/Transforms/Utils/LoopUtils.cpp b/llvm/lib/Transforms/Utils/LoopUtils.cpp
index a2e544801b9c4..5c1a6fbaebbdd 100644
--- a/llvm/lib/Transforms/Utils/LoopUtils.cpp
+++ b/llvm/lib/Transforms/Utils/LoopUtils.cpp
@@ -1790,6 +1790,10 @@ static bool hasHardUserWithinLoop(const Loop *L, const Instruction *I) {
return false;
}
+// Exit values of an instruction's loop-variant operands, indexed by operand
+// number.
+using OperandExitValueList = SmallVector<std::pair<unsigned, SCEVUse>, 2>;
+
// Collect information about PHI nodes which can be transformed in
// rewriteLoopExitValues.
struct RewritePhi {
@@ -1798,11 +1802,14 @@ struct RewritePhi {
SCEVUse ExpansionSCEV; // The SCEV of the incoming value we are rewriting.
Instruction *ExpansionPoint; // Where we'd like to expand that SCEV?
bool HighCost; // Is this expansion a high-cost?
+ // If SCEV cannot model the incoming instruction, use these values to clone
+ // the instruction with its loop-variant operands rewritten.
+ OperandExitValueList OperandExitValues;
RewritePhi(PHINode *P, unsigned I, SCEVUse Val, Instruction *ExpansionPt,
- bool H)
+ bool H, OperandExitValueList OperandExitValues)
: PN(P), Ith(I), ExpansionSCEV(Val), ExpansionPoint(ExpansionPt),
- HighCost(H) {}
+ HighCost(H), OperandExitValues(std::move(OperandExitValues)) {}
};
// Check whether it is possible to delete the loop after rewriting exit
@@ -1873,6 +1880,29 @@ static bool checkIsIndPhi(PHINode *Phi, Loop *L, ScalarEvolution *SE,
return InductionDescriptor::isInductionPHI(Phi, L, SE, ID);
}
+static bool collectOperandExitValues(Instruction *Inst, Loop *L,
+ BasicBlock *Preheader, ScalarEvolution *SE,
+ SCEVExpander &Rewriter,
+ OperandExitValueList &ExitValues) {
+ if (!Preheader || !isa<ICmpInst>(Inst))
+ return false;
+
+ for (const auto &[Idx, Op] : enumerate(Inst->operands())) {
+ // Operands defined outside the loop already dominate the preheader.
+ if (L->isLoopInvariant(Op))
+ continue;
+ if (!SE->isSCEVable(Op->getType()))
+ return false;
+ SCEVUse ExitValue = SE->getSCEVAtScope(Op, L->getParentLoop());
+ if (isa<SCEVCouldNotCompute>(ExitValue) ||
+ !SE->isLoopInvariant(ExitValue, L) ||
+ !Rewriter.isSafeToExpandAt(ExitValue, Preheader->getTerminator()))
+ return false;
+ ExitValues.emplace_back(Idx, ExitValue);
+ }
+ return !ExitValues.empty();
+}
+
int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
ScalarEvolution *SE,
const TargetTransformInfo *TTI,
@@ -1971,6 +2001,7 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
// expression reuse by the SCEVExpander), but resort to per-exit
// evaluation if that fails.
SCEVUse ExitValue = SE->getSCEVAtScope(Inst, L->getParentLoop());
+ OperandExitValueList OperandExitValues;
if (isa<SCEVCouldNotCompute>(ExitValue) ||
!SE->isLoopInvariant(ExitValue, L) ||
!Rewriter.isSafeToExpand(ExitValue)) {
@@ -1979,14 +2010,15 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
// most SCEV expressions and other recurrence types (e.g. shift
// recurrences). Is there existing code we can reuse?
const SCEV *ExitCount = SE->getExitCount(L, PN->getIncomingBlock(i));
- if (isa<SCEVCouldNotCompute>(ExitCount))
- continue;
- if (auto *AddRec = dyn_cast<SCEVAddRecExpr>(SE->getSCEV(Inst)))
- if (AddRec->getLoop() == L)
- ExitValue = AddRec->evaluateAtIteration(ExitCount, *SE);
- if (isa<SCEVCouldNotCompute>(ExitValue) ||
- !SE->isLoopInvariant(ExitValue, L) ||
- !Rewriter.isSafeToExpand(ExitValue))
+ if (!isa<SCEVCouldNotCompute>(ExitCount))
+ if (auto *AddRec = dyn_cast<SCEVAddRecExpr>(SE->getSCEV(Inst)))
+ if (AddRec->getLoop() == L)
+ ExitValue = AddRec->evaluateAtIteration(ExitCount, *SE);
+ if ((isa<SCEVCouldNotCompute>(ExitValue) ||
+ !SE->isLoopInvariant(ExitValue, L) ||
+ !Rewriter.isSafeToExpand(ExitValue)) &&
+ !collectOperandExitValues(Inst, L, L->getLoopPreheader(), SE,
+ Rewriter, OperandExitValues))
continue;
}
@@ -2000,8 +2032,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
continue;
// Check if expansions of this SCEV would count as being high cost.
- bool HighCost = Rewriter.isHighCostExpansion(
- ExitValue.getPointer(), L, SCEVCheapExpansionBudget, TTI, Inst);
+ bool HighCost =
+ OperandExitValues.empty() &&
+ Rewriter.isHighCostExpansion(ExitValue.getPointer(), L,
+ SCEVCheapExpansionBudget, TTI, Inst);
// Note that we must not perform expansions until after
// we query *all* the costs, because if we perform temporary expansion
@@ -2013,7 +2047,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
Instruction *InsertPt =
(isa<PHINode>(Inst) || isa<LandingPadInst>(Inst)) ?
&*Inst->getParent()->getFirstInsertionPt() : Inst;
- RewritePhiSet.emplace_back(PN, i, ExitValue, InsertPt, HighCost);
+ if (!OperandExitValues.empty())
+ InsertPt = L->getLoopPreheader()->getTerminator();
+ RewritePhiSet.emplace_back(PN, i, ExitValue, InsertPt, HighCost,
+ std::move(OperandExitValues));
}
}
}
@@ -2030,6 +2067,10 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
for (const RewritePhi &Phi : RewritePhiSet) {
PHINode *PN = Phi.PN;
+ // Only clone an instruction when doing so allows the loop to be deleted.
+ if (!LoopCanBeDel && !Phi.OperandExitValues.empty())
+ continue;
+
// Only do the rewrite when the ExitValue can be expanded cheaply.
// If LoopCanBeDel is true, rewrite exit value aggressively.
if ((ReplaceExitValue == OnlyCheapRepl ||
@@ -2037,12 +2078,25 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
!LoopCanBeDel && Phi.HighCost)
continue;
- Value *ExitVal = Rewriter.expandCodeFor(
- Phi.ExpansionSCEV, Phi.PN->getType(), Phi.ExpansionPoint);
+ Instruction *Inst = cast<Instruction>(PN->getIncomingValue(Phi.Ith));
+ Value *ExitVal;
+ if (Phi.OperandExitValues.empty()) {
+ ExitVal = Rewriter.expandCodeFor(Phi.ExpansionSCEV, Phi.PN->getType(),
+ Phi.ExpansionPoint);
+ } else {
+ Instruction *Clone = Inst->clone();
+ Clone->setName(Inst->getName() + ".exit");
+ for (auto [Idx, ExitSCEV] : Phi.OperandExitValues)
+ Clone->setOperand(Idx, Rewriter.expandCodeFor(
+ ExitSCEV, Inst->getOperand(Idx)->getType(),
+ Phi.ExpansionPoint));
+ Clone->insertBefore(Phi.ExpansionPoint->getIterator());
+ ExitVal = Clone;
+ }
LLVM_DEBUG(dbgs() << "rewriteLoopExitValues: AfterLoopVal = " << *ExitVal
<< '\n'
- << " LoopVal = " << *(Phi.ExpansionPoint) << "\n");
+ << " LoopVal = " << *Inst << "\n");
#ifndef NDEBUG
// If we reuse an instruction from a loop which is neither L nor one of
@@ -2055,7 +2109,6 @@ int llvm::rewriteLoopExitValues(Loop *L, LoopInfo *LI, TargetLibraryInfo *TLI,
#endif
NumReplaced++;
- Instruction *Inst = cast<Instruction>(PN->getIncomingValue(Phi.Ith));
PN->setIncomingValue(Phi.Ith, ExitVal);
// It's necessary to tell ScalarEvolution about this explicitly so that
// it can walk the def-use list and forget all SCEVs, as it may not be
diff --git a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
index 09e8765610f69..d9677d69e48d5 100644
--- a/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
+++ b/llvm/test/Transforms/IndVarSimplify/rewrite-loop-exit-value.ll
@@ -309,17 +309,15 @@ define i1 @rewrite_computable_icmp(i32 %start, i32 %limit) {
; CHECK-NEXT: entry:
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT: [[CMP_EXIT:%.*]] = icmp ne i32 [[IV]], 42
-; CHECK-NEXT: [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT: [[CONTINUE:%.*]] = select i1 [[CMP_EXIT]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT: br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT: br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
; CHECK: body:
-; CHECK-NEXT: [[IV_NEXT]] = add nsw i32 [[IV]], -1
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
; CHECK-NEXT: br label [[LOOP]]
; CHECK: exit:
+; CHECK-NEXT: [[TMP0:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT: [[TMP1:%.*]] = add i32 [[START:%.*]], -42
+; CHECK-NEXT: [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[TMP0]], i32 [[TMP1]])
+; CHECK-NEXT: [[TMP2:%.*]] = sub i32 [[START]], [[UMIN]]
+; CHECK-NEXT: [[CMP_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
; CHECK-NEXT: ret i1 [[CMP_EXIT]]
;
entry:
@@ -348,19 +346,17 @@ define i1 @rewrite_multiple_icmps(i32 %start, i32 %rhs.start, i32 %limit) {
; CHECK-NEXT: entry:
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[TMP2:%.*]] = phi i32 [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[IV_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT: [[TMP4:%.*]] = phi i32 [ [[RHS_START:%.*]], [[ENTRY]] ], [ [[RHS_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT: [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
-; CHECK-NEXT: [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT: [[CONTINUE:%.*]] = select i1 [[CMP0_EXIT]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT: br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT: br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
; CHECK: body:
-; CHECK-NEXT: [[IV_NEXT]] = add nsw i32 [[TMP2]], -1
-; CHECK-NEXT: [[RHS_NEXT]] = add i32 [[TMP4]], 2
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
; CHECK-NEXT: br label [[LOOP]]
; CHECK: exit:
+; CHECK-NEXT: [[TMP0:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT: [[TMP1:%.*]] = add i32 [[START:%.*]], -42
+; CHECK-NEXT: [[UMIN:%.*]] = call i32 @llvm.umin.i32(i32 [[TMP0]], i32 [[TMP1]])
+; CHECK-NEXT: [[TMP2:%.*]] = sub i32 [[START]], [[UMIN]]
+; CHECK-NEXT: [[CMP0_EXIT:%.*]] = icmp ne i32 [[TMP2]], 42
+; CHECK-NEXT: [[TMP3:%.*]] = shl i32 [[UMIN]], 1
+; CHECK-NEXT: [[TMP4:%.*]] = add i32 [[RHS_START:%.*]], [[TMP3]]
; CHECK-NEXT: [[CMP1_EXIT:%.*]] = icmp sgt i32 [[TMP2]], [[TMP4]]
; CHECK-NEXT: [[RESULT:%.*]] = and i1 [[CMP0_EXIT]], [[CMP1_EXIT]]
; CHECK-NEXT: ret i1 [[RESULT]]
@@ -462,17 +458,18 @@ define i1 @rewrite_pointer_icmp(ptr %start, ptr %end, i32 %limit) {
; CHECK-NEXT: entry:
; CHECK-NEXT: br label [[LOOP:%.*]]
; CHECK: loop:
-; CHECK-NEXT: [[PTR:%.*]] = phi ptr [ [[START:%.*]], [[ENTRY:%.*]] ], [ [[PTR_NEXT:%.*]], [[BODY:%.*]] ]
-; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, [[ENTRY]] ], [ [[INDEX_NEXT:%.*]], [[BODY]] ]
-; CHECK-NEXT: [[CMP:%.*]] = icmp ne ptr [[PTR]], [[END:%.*]]
-; CHECK-NEXT: [[INRANGE:%.*]] = icmp ult i32 [[INDEX]], [[LIMIT:%.*]]
-; CHECK-NEXT: [[CONTINUE:%.*]] = select i1 [[CMP]], i1 [[INRANGE]], i1 false
-; CHECK-NEXT: br i1 [[CONTINUE]], label [[BODY]], label [[EXIT:%.*]]
+; CHECK-NEXT: br i1 false, label [[BODY:%.*]], label [[EXIT:%.*]]
; CHECK: body:
-; CHECK-NEXT: [[PTR_NEXT]] = getelementptr i8, ptr [[PTR]], i64 1
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 1
; CHECK-NEXT: br label [[LOOP]]
; CHECK: exit:
+; CHECK-NEXT: [[START2:%.*]] = ptrtoaddr ptr [[START:%.*]] to i64
+; CHECK-NEXT: [[END1:%.*]] = ptrtoaddr ptr [[END:%.*]] to i64
+; CHECK-NEXT: [[LIMIT_FR:%.*]] = freeze i32 [[LIMIT:%.*]]
+; CHECK-NEXT: [[TMP0:%.*]] = zext i32 [[LIMIT_FR]] to i64
+; CHECK-NEXT: [[TMP1:%.*]] = sub i64 [[END1]], [[START2]]
+; CHECK-NEXT: [[UMIN:%.*]] = call i64 @llvm.umin.i64(i64 [[TMP0]], i64 [[TMP1]])
+; CHECK-NEXT: [[SCEVGEP:%.*]] = getelementptr i8, ptr [[START]], i64 [[UMIN]]
+; CHECK-NEXT: [[CMP:%.*]] = icmp ne ptr [[SCEVGEP]], [[END]]
; CHECK-NEXT: ret i1 [[CMP]]
;
entry:
diff --git a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
index f91972263d862..86f128debdb31 100644
--- a/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
+++ b/llvm/test/Transforms/PhaseOrdering/RISCV/rewrite-loop-exit-compare.ll
@@ -13,35 +13,15 @@ define i1 @rewrite_loop_exit_compare(i32 %length, i64 %limit) vscale_range(2, 10
; CHECK-NEXT: [[CONTINUE3:%.*]] = and i1 [[NONZERO1]], [[INRANGE2]]
; CHECK-NEXT: br i1 [[CONTINUE3]], label %[[BODY_PREHEADER:.*]], label %[[EXIT:.*]]
; CHECK: [[BODY_PREHEADER]]:
-; CHECK-NEXT: [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
; CHECK-NEXT: [[TMP0:%.*]] = add i32 [[LENGTH]], -1
+; CHECK-NEXT: [[TMP1:%.*]] = add i64 [[LIMIT_FR]], -1
; CHECK-NEXT: [[TMP2:%.*]] = zext i32 [[TMP0]] to i64
; CHECK-NEXT: [[UMIN:%.*]] = tail call i64 @llvm.umin.i64(i64 [[TMP1]], i64 [[TMP2]])
-; CHECK-NEXT: [[TMP4:%.*]] = add nuw nsw i64 [[UMIN]], 1
-; CHECK-NEXT: [[TMP5:%.*]] = tail call <vscale x 16 x i32> @llvm.stepvector.nxv16i32()
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[LENGTH]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP6:%.*]] = sub nsw <vscale x 16 x i32> [[BROADCAST_SPLAT]], [[TMP5]]
-; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
-; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <vscale x 16 x i32> [ [[TMP6]], %[[BODY_PREHEADER]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[AVL:%.*]] = phi i64 [ [[TMP4]], %[[BODY_PREHEADER]] ], [ [[AVL_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP7:%.*]] = tail call i32 @llvm.experimental.get.vector.length.i64(i64 [[AVL]], i32 16, i1 true)
-; CHECK-NEXT: [[TMP8:%.*]] = zext i32 [[TMP7]] to i64
-; CHECK-NEXT: [[TMP9:%.*]] = sub nsw i32 0, [[TMP7]]
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT9:%.*]] = insertelement <vscale x 16 x i32> poison, i32 [[TMP9]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT10:%.*]] = shufflevector <vscale x 16 x i32> [[BROADCAST_SPLATINSERT9]], <vscale x 16 x i32> poison, <vscale x 16 x i32> zeroinitializer
-; CHECK-NEXT: [[AVL_NEXT]] = sub nuw i64 [[AVL]], [[TMP8]]
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add nsw <vscale x 16 x i32> [[VEC_IND]], [[BROADCAST_SPLAT10]]
-; CHECK-NEXT: [[TMP10:%.*]] = icmp eq i64 [[AVL_NEXT]], 0
-; CHECK-NEXT: br i1 [[TMP10]], label %[[EXIT_LOOPEXIT:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
-; CHECK: [[EXIT_LOOPEXIT]]:
-; CHECK-NEXT: [[TMP11:%.*]] = add nsw i64 [[TMP8]], -1
-; CHECK-NEXT: [[TMP12:%.*]] = extractelement <vscale x 16 x i32> [[VEC_IND]], i64 [[TMP11]]
-; CHECK-NEXT: [[TMP13:%.*]] = icmp ne i32 [[TMP12]], 1
+; CHECK-NEXT: [[TMP3:%.*]] = trunc nuw i64 [[UMIN]] to i32
+; CHECK-NEXT: [[NONZERO_EXIT:%.*]] = icmp ne i32 [[TMP0]], [[TMP3]]
; CHECK-NEXT: br label %[[EXIT]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[TMP13]], %[[EXIT_LOOPEXIT]] ]
+; CHECK-NEXT: [[NONZERO_LCSSA:%.*]] = phi i1 [ [[NONZERO1]], %[[ENTRY]] ], [ [[NONZERO_EXIT]], %[[BODY_PREHEADER]] ]
; CHECK-NEXT: ret i1 [[NONZERO_LCSSA]]
;
entry:
@@ -63,8 +43,3 @@ body:
exit:
ret i1 %nonzero
}
-;.
-; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
-; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
-; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
-;.
More information about the llvm-commits
mailing list