[llvm] [LoopVectorize] Fix double-application of FindIV reduction expression in epilogue (PR #219362)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 23:00:39 PDT 2026
https://github.com/im-lunex updated https://github.com/llvm/llvm-project/pull/219362
>From 57147bd15d2ebc3d938f8ab149e231b4cdfa16a0 Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Fri, 28 Aug 2026 10:25:04 +0600
Subject: [PATCH] [LoopVectorize] Fix double-application in epilogue
---
.../Vectorize/LoopVectorizationPlanner.cpp | 14 +----
llvm/lib/Transforms/Vectorize/VPlan.h | 10 +++-
.../Transforms/Vectorize/VPlanTransforms.cpp | 7 ++-
.../X86/find-iv-sunk-expr-epilogue.ll | 52 +++++++++++++++++++
4 files changed, 69 insertions(+), 14 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index c464c7894ae5c..82634ffb47b9d 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -879,18 +879,8 @@ static bool hasUnsupportedHeaderPhiRecipe(VPlan &Plan) {
RecurrenceDescriptor::isFindLastRecurrenceKind(Kind) ||
!RedPhi->getUnderlyingValue())
return true;
- // TODO: Add support for FindIV reductions with sunk expressions: the
- // resume value from the main loop is in expression domain (e.g.,
- // mul(ReducedIV, 3)), but the epilogue tracks raw IV values. A sunk
- // expression is identified by a non-VPInstruction user of
- // ComputeReductionResult.
- if (RecurrenceDescriptor::isFindIVRecurrenceKind(Kind)) {
- auto *RdxResult = vputils::findComputeReductionResult(RedPhi);
- assert(RdxResult &&
- "FindIV reduction must have ComputeReductionResult");
- return any_of(RdxResult->users(),
- std::not_fn(IsaPred<VPInstruction>));
- }
+ if (RecurrenceDescriptor::isFindIVRecurrenceKind(Kind))
+ return RedPhi->isExpressionSunk();
return false;
}
default:
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 7d2c2fa1bdd23..aaa456ee2e435 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -2879,6 +2879,8 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
/// compare has multiple uses.
bool HasUsesOutsideReductionChain;
+ bool ExpressionSunk = false;
+
public:
/// Create a new VPReductionPHIRecipe for the reduction \p Phi.
VPReductionPHIRecipe(PHINode *Phi, RecurKind Kind, VPValue &Start,
@@ -2895,9 +2897,11 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
VPReductionPHIRecipe *cloneWithOperands(VPValue *Start,
VPValue *BackedgeValue) {
- return new VPReductionPHIRecipe(
+ auto *Clone = new VPReductionPHIRecipe(
dyn_cast_or_null<PHINode>(getUnderlyingValue()), getRecurrenceKind(),
*Start, *BackedgeValue, Style, *this, HasUsesOutsideReductionChain);
+ Clone->ExpressionSunk = ExpressionSunk;
+ return Clone;
}
VPReductionPHIRecipe *clone() override {
@@ -2943,6 +2947,10 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
return HasUsesOutsideReductionChain;
}
+ void setExpressionSunk(bool V = true) { ExpressionSunk = V; }
+
+ bool isExpressionSunk() const { return ExpressionSunk; }
+
/// Returns true if the recipe only uses the first lane of operand \p Op.
bool usesFirstLaneOnly(const VPValue *Op) const override {
assert(is_contained(operands(), Op) &&
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index dd85eeaa8592c..a75af19cab500 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4573,10 +4573,13 @@ void VPlanTransforms::optimizeFindIVReductions(VPlan &Plan,
// If IVOfExpressionToSink is an expression to sink, sink it now.
VPValue *VectorRegionExitingVal = ReducedIV;
- if (IVOfExpressionToSink)
+ bool SunkExpression = false;
+ if (IVOfExpressionToSink) {
VectorRegionExitingVal =
cloneBinOpForScalarIV(cast<VPWidenRecipe>(FindLastExpression),
ReducedIV, IVOfExpressionToSink);
+ SunkExpression = true;
+ }
VPValue *NewRdxResult;
VPValue *StartVPV = PhiR->getStartValue();
@@ -4617,6 +4620,8 @@ void VPlanTransforms::optimizeFindIVReductions(VPlan &Plan,
cast<PHINode>(PhiR->getUnderlyingInstr()), RecurKind::FindIV, *StartVPV,
*NewFindLastSelect, RdxUnordered{1}, {},
PhiR->hasUsesOutsideReductionChain());
+ if (SunkExpression)
+ NewPhiR->setExpressionSunk();
NewPhiR->insertBefore(PhiR);
PhiR->replaceAllUsesWith(NewPhiR);
PhiR->eraseFromParent();
diff --git a/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll b/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
new file mode 100644
index 0000000000000..db46f9ee616d2
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
@@ -0,0 +1,52 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -force-vector-interleave=1 -force-vector-width=4 -enable-epilogue-vectorization -epilogue-vectorization-force-VF=4 -S %s | FileCheck %s
+
+; Test for https://github.com/llvm/llvm-project/issues/219211
+
+; CHECK-LABEL: define i32 @findiv_mul_pow2_sunk(
+; CHECK: vector.body:
+; CHECK: middle.block:
+; CHECK: shl i32 {{.*}}, 2
+; CHECK-NOT: vec.epilog
+; CHECK: ret i32
+define i32 @findiv_mul_pow2_sunk(ptr %a, i32 %n) #0 {
+entry:
+ br label %loop
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+ %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+ %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+ %l = load i32, ptr %gep, align 4
+ %c = icmp eq i32 %l, 42
+ %expr = mul i32 %iv, 4
+ %sel = select i1 %c, i32 %expr, i32 %rdx
+ %iv.next = add nuw nsw i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, %n
+ br i1 %ec, label %done, label %loop
+done:
+ ret i32 %sel
+}
+
+; CHECK-LABEL: define i32 @findiv_no_sunk_raw_iv(
+; CHECK: vector.body:
+; CHECK: middle.block:
+; CHECK-NOT: shl i32
+; CHECK: ret i32
+define i32 @findiv_no_sunk_raw_iv(ptr %a, i32 %n) #0 {
+entry:
+ br label %loop
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+ %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+ %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+ %l = load i32, ptr %gep, align 4
+ %c = icmp eq i32 %l, 42
+ %sel = select i1 %c, i32 %iv, i32 %rdx
+ %iv.next = add nuw nsw i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, %n
+ br i1 %ec, label %done, label %loop
+done:
+ ret i32 %sel
+}
+
+attributes #0 = { "target-features"="+avx512f" }
More information about the llvm-commits
mailing list