[llvm] [LoopVectorize] Fix double-application of FindIV reduction expression in epilogue (PR #219362)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 20:55:55 PDT 2026
https://github.com/im-lunex updated https://github.com/llvm/llvm-project/pull/219362
>From 57147bd15d2ebc3d938f8ab149e231b4cdfa16a0 Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Fri, 28 Aug 2026 10:25:04 +0600
Subject: [PATCH 1/4] [LoopVectorize] Fix double-application in epilogue
---
.../Vectorize/LoopVectorizationPlanner.cpp | 14 +----
llvm/lib/Transforms/Vectorize/VPlan.h | 10 +++-
.../Transforms/Vectorize/VPlanTransforms.cpp | 7 ++-
.../X86/find-iv-sunk-expr-epilogue.ll | 52 +++++++++++++++++++
4 files changed, 69 insertions(+), 14 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index c464c7894ae5c..82634ffb47b9d 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -879,18 +879,8 @@ static bool hasUnsupportedHeaderPhiRecipe(VPlan &Plan) {
RecurrenceDescriptor::isFindLastRecurrenceKind(Kind) ||
!RedPhi->getUnderlyingValue())
return true;
- // TODO: Add support for FindIV reductions with sunk expressions: the
- // resume value from the main loop is in expression domain (e.g.,
- // mul(ReducedIV, 3)), but the epilogue tracks raw IV values. A sunk
- // expression is identified by a non-VPInstruction user of
- // ComputeReductionResult.
- if (RecurrenceDescriptor::isFindIVRecurrenceKind(Kind)) {
- auto *RdxResult = vputils::findComputeReductionResult(RedPhi);
- assert(RdxResult &&
- "FindIV reduction must have ComputeReductionResult");
- return any_of(RdxResult->users(),
- std::not_fn(IsaPred<VPInstruction>));
- }
+ if (RecurrenceDescriptor::isFindIVRecurrenceKind(Kind))
+ return RedPhi->isExpressionSunk();
return false;
}
default:
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 7d2c2fa1bdd23..aaa456ee2e435 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -2879,6 +2879,8 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
/// compare has multiple uses.
bool HasUsesOutsideReductionChain;
+ bool ExpressionSunk = false;
+
public:
/// Create a new VPReductionPHIRecipe for the reduction \p Phi.
VPReductionPHIRecipe(PHINode *Phi, RecurKind Kind, VPValue &Start,
@@ -2895,9 +2897,11 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
VPReductionPHIRecipe *cloneWithOperands(VPValue *Start,
VPValue *BackedgeValue) {
- return new VPReductionPHIRecipe(
+ auto *Clone = new VPReductionPHIRecipe(
dyn_cast_or_null<PHINode>(getUnderlyingValue()), getRecurrenceKind(),
*Start, *BackedgeValue, Style, *this, HasUsesOutsideReductionChain);
+ Clone->ExpressionSunk = ExpressionSunk;
+ return Clone;
}
VPReductionPHIRecipe *clone() override {
@@ -2943,6 +2947,10 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
return HasUsesOutsideReductionChain;
}
+ void setExpressionSunk(bool V = true) { ExpressionSunk = V; }
+
+ bool isExpressionSunk() const { return ExpressionSunk; }
+
/// Returns true if the recipe only uses the first lane of operand \p Op.
bool usesFirstLaneOnly(const VPValue *Op) const override {
assert(is_contained(operands(), Op) &&
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index dd85eeaa8592c..a75af19cab500 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4573,10 +4573,13 @@ void VPlanTransforms::optimizeFindIVReductions(VPlan &Plan,
// If IVOfExpressionToSink is an expression to sink, sink it now.
VPValue *VectorRegionExitingVal = ReducedIV;
- if (IVOfExpressionToSink)
+ bool SunkExpression = false;
+ if (IVOfExpressionToSink) {
VectorRegionExitingVal =
cloneBinOpForScalarIV(cast<VPWidenRecipe>(FindLastExpression),
ReducedIV, IVOfExpressionToSink);
+ SunkExpression = true;
+ }
VPValue *NewRdxResult;
VPValue *StartVPV = PhiR->getStartValue();
@@ -4617,6 +4620,8 @@ void VPlanTransforms::optimizeFindIVReductions(VPlan &Plan,
cast<PHINode>(PhiR->getUnderlyingInstr()), RecurKind::FindIV, *StartVPV,
*NewFindLastSelect, RdxUnordered{1}, {},
PhiR->hasUsesOutsideReductionChain());
+ if (SunkExpression)
+ NewPhiR->setExpressionSunk();
NewPhiR->insertBefore(PhiR);
PhiR->replaceAllUsesWith(NewPhiR);
PhiR->eraseFromParent();
diff --git a/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll b/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
new file mode 100644
index 0000000000000..db46f9ee616d2
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
@@ -0,0 +1,52 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -force-vector-interleave=1 -force-vector-width=4 -enable-epilogue-vectorization -epilogue-vectorization-force-VF=4 -S %s | FileCheck %s
+
+; Test for https://github.com/llvm/llvm-project/issues/219211
+
+; CHECK-LABEL: define i32 @findiv_mul_pow2_sunk(
+; CHECK: vector.body:
+; CHECK: middle.block:
+; CHECK: shl i32 {{.*}}, 2
+; CHECK-NOT: vec.epilog
+; CHECK: ret i32
+define i32 @findiv_mul_pow2_sunk(ptr %a, i32 %n) #0 {
+entry:
+ br label %loop
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+ %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+ %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+ %l = load i32, ptr %gep, align 4
+ %c = icmp eq i32 %l, 42
+ %expr = mul i32 %iv, 4
+ %sel = select i1 %c, i32 %expr, i32 %rdx
+ %iv.next = add nuw nsw i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, %n
+ br i1 %ec, label %done, label %loop
+done:
+ ret i32 %sel
+}
+
+; CHECK-LABEL: define i32 @findiv_no_sunk_raw_iv(
+; CHECK: vector.body:
+; CHECK: middle.block:
+; CHECK-NOT: shl i32
+; CHECK: ret i32
+define i32 @findiv_no_sunk_raw_iv(ptr %a, i32 %n) #0 {
+entry:
+ br label %loop
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+ %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+ %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+ %l = load i32, ptr %gep, align 4
+ %c = icmp eq i32 %l, 42
+ %sel = select i1 %c, i32 %iv, i32 %rdx
+ %iv.next = add nuw nsw i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, %n
+ br i1 %ec, label %done, label %loop
+done:
+ ret i32 %sel
+}
+
+attributes #0 = { "target-features"="+avx512f" }
>From 20b59cb7928a7695b0395222d52ee98722255a2e Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Wed, 2 Sep 2026 00:13:34 +0600
Subject: [PATCH 2/4] rewrite test and regenerate CHECK's
---
.../Vectorize/LoopVectorizationPlanner.cpp | 5 +
llvm/lib/Transforms/Vectorize/VPlan.h | 5 +-
.../X86/find-iv-sunk-expr-epilogue.ll | 52 -------
.../find-iv-sunk-expr-epilogue.ll | 131 ++++++++++++++++++
4 files changed, 140 insertions(+), 53 deletions(-)
delete mode 100644 llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
create mode 100644 llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index 82634ffb47b9d..0926781a02902 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -879,6 +879,11 @@ static bool hasUnsupportedHeaderPhiRecipe(VPlan &Plan) {
RecurrenceDescriptor::isFindLastRecurrenceKind(Kind) ||
!RedPhi->getUnderlyingValue())
return true;
+ // TODO: Add support for FindIV reductions with sunk expressions: the
+ // resume value from the main loop is in expression domain (e.g.,
+ // mul(ReducedIV, 3)), but the epilogue tracks raw IV values. A sunk
+ // expression is identified by a non-VPInstruction user of
+ // ComputeReductionResult.
if (RecurrenceDescriptor::isFindIVRecurrenceKind(Kind))
return RedPhi->isExpressionSunk();
return false;
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index aaa456ee2e435..9d0fbb182a255 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -2879,6 +2879,9 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
/// compare has multiple uses.
bool HasUsesOutsideReductionChain;
+ // True if FindIV's expression was sunk into the vector loop. Epilogue
+ // is disabled while this is true. Temporary until epilogue handles sunk
+ // expressions.
bool ExpressionSunk = false;
public:
@@ -2947,7 +2950,7 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
return HasUsesOutsideReductionChain;
}
- void setExpressionSunk(bool V = true) { ExpressionSunk = V; }
+ void setExpressionSunk() { ExpressionSunk = true; }
bool isExpressionSunk() const { return ExpressionSunk; }
diff --git a/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll b/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
deleted file mode 100644
index db46f9ee616d2..0000000000000
--- a/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
+++ /dev/null
@@ -1,52 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; RUN: opt -passes=loop-vectorize -force-vector-interleave=1 -force-vector-width=4 -enable-epilogue-vectorization -epilogue-vectorization-force-VF=4 -S %s | FileCheck %s
-
-; Test for https://github.com/llvm/llvm-project/issues/219211
-
-; CHECK-LABEL: define i32 @findiv_mul_pow2_sunk(
-; CHECK: vector.body:
-; CHECK: middle.block:
-; CHECK: shl i32 {{.*}}, 2
-; CHECK-NOT: vec.epilog
-; CHECK: ret i32
-define i32 @findiv_mul_pow2_sunk(ptr %a, i32 %n) #0 {
-entry:
- br label %loop
-loop:
- %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
- %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
- %gep = getelementptr inbounds i32, ptr %a, i32 %iv
- %l = load i32, ptr %gep, align 4
- %c = icmp eq i32 %l, 42
- %expr = mul i32 %iv, 4
- %sel = select i1 %c, i32 %expr, i32 %rdx
- %iv.next = add nuw nsw i32 %iv, 1
- %ec = icmp eq i32 %iv.next, %n
- br i1 %ec, label %done, label %loop
-done:
- ret i32 %sel
-}
-
-; CHECK-LABEL: define i32 @findiv_no_sunk_raw_iv(
-; CHECK: vector.body:
-; CHECK: middle.block:
-; CHECK-NOT: shl i32
-; CHECK: ret i32
-define i32 @findiv_no_sunk_raw_iv(ptr %a, i32 %n) #0 {
-entry:
- br label %loop
-loop:
- %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
- %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
- %gep = getelementptr inbounds i32, ptr %a, i32 %iv
- %l = load i32, ptr %gep, align 4
- %c = icmp eq i32 %l, 42
- %sel = select i1 %c, i32 %iv, i32 %rdx
- %iv.next = add nuw nsw i32 %iv, 1
- %ec = icmp eq i32 %iv.next, %n
- br i1 %ec, label %done, label %loop
-done:
- ret i32 %sel
-}
-
-attributes #0 = { "target-features"="+avx512f" }
diff --git a/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll b/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll
new file mode 100644
index 0000000000000..ba2a91f97e274
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll
@@ -0,0 +1,131 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -force-vector-interleave=1 -force-vector-width=4 -enable-epilogue-vectorization -epilogue-vectorization-force-VF=4 -S %s | FileCheck %s
+
+; Test for https://github.com/llvm/llvm-project/issues/219211
+
+define i32 @findiv_mul_pow2_sunk(ptr %a, i32 %n) {
+; CHECK-LABEL: define i32 @findiv_mul_pow2_sunk(
+; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[N]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP0:%.*]] = and i32 [[N]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[N]], [[TMP0]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i32> [ splat (i32 -2147483648), %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = icmp eq <4 x i32> [[WIDE_LOAD]], splat (i32 42)
+; CHECK-NEXT: [[TMP3]] = select <4 x i1> [[TMP2]], <4 x i32> [[VEC_IND]], <4 x i32> [[VEC_PHI]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i32> [[VEC_IND]], splat (i32 4)
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP5:%.*]] = call i32 @llvm.vector.reduce.smax.v4i32(<4 x i32> [[TMP3]])
+; CHECK-NEXT: [[TMP6:%.*]] = shl i32 [[TMP5]], 2
+; CHECK-NEXT: [[TMP7:%.*]] = icmp ne i32 [[TMP5]], -2147483648
+; CHECK-NEXT: [[TMP8:%.*]] = select i1 [[TMP7]], i32 [[TMP6]], i32 -1
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[DONE:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP8]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[RDX:%.*]] = phi i32 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[SEL:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[IV]]
+; CHECK-NEXT: [[L:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT: [[C:%.*]] = icmp eq i32 [[L]], 42
+; CHECK-NEXT: [[EXPR:%.*]] = mul i32 [[IV]], 4
+; CHECK-NEXT: [[SEL]] = select i1 [[C]], i32 [[EXPR]], i32 [[RDX]]
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i32 [[IV]], 1
+; CHECK-NEXT: [[EC:%.*]] = icmp eq i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[EC]], label %[[DONE]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK: [[DONE]]:
+; CHECK-NEXT: [[SEL_LCSSA:%.*]] = phi i32 [ [[SEL]], %[[LOOP]] ], [ [[TMP8]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i32 [[SEL_LCSSA]]
+;
+entry:
+ br label %loop
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+ %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+ %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+ %l = load i32, ptr %gep, align 4
+ %c = icmp eq i32 %l, 42
+ %expr = mul i32 %iv, 4
+ %sel = select i1 %c, i32 %expr, i32 %rdx
+ %iv.next = add nuw nsw i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, %n
+ br i1 %ec, label %done, label %loop
+done:
+ ret i32 %sel
+}
+
+define i32 @findiv_no_sunk_raw_iv(ptr %a, i32 %n) {
+; CHECK-LABEL: define i32 @findiv_no_sunk_raw_iv(
+; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*]]:
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[N]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP0:%.*]] = and i32 [[N]], 3
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[N]], [[TMP0]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i32> [ splat (i32 -2147483648), %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[INDEX]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT: [[TMP2:%.*]] = icmp eq <4 x i32> [[WIDE_LOAD]], splat (i32 42)
+; CHECK-NEXT: [[TMP3]] = select <4 x i1> [[TMP2]], <4 x i32> [[VEC_IND]], <4 x i32> [[VEC_PHI]]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add nuw nsw <4 x i32> [[VEC_IND]], splat (i32 4)
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP5:%.*]] = call i32 @llvm.vector.reduce.smax.v4i32(<4 x i32> [[TMP3]])
+; CHECK-NEXT: [[TMP6:%.*]] = icmp ne i32 [[TMP5]], -2147483648
+; CHECK-NEXT: [[TMP7:%.*]] = select i1 [[TMP6]], i32 [[TMP5]], i32 -1
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[N]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[DONE:.*]], label %[[SCALAR_PH]]
+; CHECK: [[SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[RDX:%.*]] = phi i32 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[SEL:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[IV]]
+; CHECK-NEXT: [[L:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT: [[C:%.*]] = icmp eq i32 [[L]], 42
+; CHECK-NEXT: [[SEL]] = select i1 [[C]], i32 [[IV]], i32 [[RDX]]
+; CHECK-NEXT: [[IV_NEXT]] = add nuw nsw i32 [[IV]], 1
+; CHECK-NEXT: [[EC:%.*]] = icmp eq i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[EC]], label %[[DONE]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[DONE]]:
+; CHECK-NEXT: [[SEL_LCSSA:%.*]] = phi i32 [ [[SEL]], %[[LOOP]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i32 [[SEL_LCSSA]]
+;
+entry:
+ br label %loop
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+ %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+ %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+ %l = load i32, ptr %gep, align 4
+ %c = icmp eq i32 %l, 42
+ %sel = select i1 %c, i32 %iv, i32 %rdx
+ %iv.next = add nuw nsw i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, %n
+ br i1 %ec, label %done, label %loop
+done:
+ ret i32 %sel
+}
>From e9db3bde05a562babae2a7d8bebb2fac2fa5d4e2 Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Wed, 23 Sep 2026 15:07:30 +0600
Subject: [PATCH 3/4] use Doxygen format instead
---
llvm/lib/Transforms/Vectorize/VPlan.h | 6 +++---
1 file changed, 3 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 9d0fbb182a255..fa9becbb4148d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -2879,9 +2879,9 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
/// compare has multiple uses.
bool HasUsesOutsideReductionChain;
- // True if FindIV's expression was sunk into the vector loop. Epilogue
- // is disabled while this is true. Temporary until epilogue handles sunk
- // expressions.
+ /// True if FindIV's expression was sunk into the vector loop. Epilogue
+ /// is disabled while this is true. Temporary until epilogue handles sunk
+ /// expressions.
bool ExpressionSunk = false;
public:
>From 8b86e3348af564460de96e985ec75fcbe70359cb Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Thu, 1 Oct 2026 09:46:59 +0600
Subject: [PATCH 4/4] improve comment and address reviews and add todo
---
llvm/lib/Transforms/Vectorize/VPlan.h | 10 +++++++---
.../LoopVectorize/find-iv-sunk-expr-epilogue.ll | 2 ++
2 files changed, 9 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index fa9becbb4148d..f9b539463a4a2 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -2879,9 +2879,13 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
/// compare has multiple uses.
bool HasUsesOutsideReductionChain;
- /// True if FindIV's expression was sunk into the vector loop. Epilogue
- /// is disabled while this is true. Temporary until epilogue handles sunk
- /// expressions.
+ /// Temporary flag indicating that the FindIV reduction expression has been
+ /// sunk. While this is true, epilogue vectorization is
+ /// disabled to avoid applying the expression twice (once in the main vector
+ /// loop and again in the epilogue), which can produce incorrect results by
+ /// doing the sunk operation twice.
+ /// TODO: Remove this flag once epilogue vectorization properly supports sunk
+ /// FindIV expressions.
bool ExpressionSunk = false;
public:
diff --git a/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll b/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll
index ba2a91f97e274..d1398022508ec 100644
--- a/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll
+++ b/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll
@@ -116,6 +116,7 @@ define i32 @findiv_no_sunk_raw_iv(ptr %a, i32 %n) {
;
entry:
br label %loop
+
loop:
%iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
%rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
@@ -126,6 +127,7 @@ loop:
%iv.next = add nuw nsw i32 %iv, 1
%ec = icmp eq i32 %iv.next, %n
br i1 %ec, label %done, label %loop
+
done:
ret i32 %sel
}
More information about the llvm-commits
mailing list