[llvm] [LoopVectorize] Fix double-application of FindIV reduction expression in epilogue (PR #219362)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 30 20:55:55 PDT 2026


https://github.com/im-lunex updated https://github.com/llvm/llvm-project/pull/219362

>From 57147bd15d2ebc3d938f8ab149e231b4cdfa16a0 Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Fri, 28 Aug 2026 10:25:04 +0600
Subject: [PATCH 1/4] [LoopVectorize] Fix double-application in epilogue

---
 .../Vectorize/LoopVectorizationPlanner.cpp    | 14 +----
 llvm/lib/Transforms/Vectorize/VPlan.h         | 10 +++-
 .../Transforms/Vectorize/VPlanTransforms.cpp  |  7 ++-
 .../X86/find-iv-sunk-expr-epilogue.ll         | 52 +++++++++++++++++++
 4 files changed, 69 insertions(+), 14 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index c464c7894ae5c..82634ffb47b9d 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -879,18 +879,8 @@ static bool hasUnsupportedHeaderPhiRecipe(VPlan &Plan) {
               RecurrenceDescriptor::isFindLastRecurrenceKind(Kind) ||
               !RedPhi->getUnderlyingValue())
             return true;
-          // TODO: Add support for FindIV reductions with sunk expressions: the
-          // resume value from the main loop is in expression domain (e.g.,
-          // mul(ReducedIV, 3)), but the epilogue tracks raw IV values. A sunk
-          // expression is identified by a non-VPInstruction user of
-          // ComputeReductionResult.
-          if (RecurrenceDescriptor::isFindIVRecurrenceKind(Kind)) {
-            auto *RdxResult = vputils::findComputeReductionResult(RedPhi);
-            assert(RdxResult &&
-                   "FindIV reduction must have ComputeReductionResult");
-            return any_of(RdxResult->users(),
-                          std::not_fn(IsaPred<VPInstruction>));
-          }
+          if (RecurrenceDescriptor::isFindIVRecurrenceKind(Kind))
+            return RedPhi->isExpressionSunk();
           return false;
         }
         default:
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 7d2c2fa1bdd23..aaa456ee2e435 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -2879,6 +2879,8 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
   /// compare has multiple uses.
   bool HasUsesOutsideReductionChain;
 
+  bool ExpressionSunk = false;
+
 public:
   /// Create a new VPReductionPHIRecipe for the reduction \p Phi.
   VPReductionPHIRecipe(PHINode *Phi, RecurKind Kind, VPValue &Start,
@@ -2895,9 +2897,11 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
 
   VPReductionPHIRecipe *cloneWithOperands(VPValue *Start,
                                           VPValue *BackedgeValue) {
-    return new VPReductionPHIRecipe(
+    auto *Clone = new VPReductionPHIRecipe(
         dyn_cast_or_null<PHINode>(getUnderlyingValue()), getRecurrenceKind(),
         *Start, *BackedgeValue, Style, *this, HasUsesOutsideReductionChain);
+    Clone->ExpressionSunk = ExpressionSunk;
+    return Clone;
   }
 
   VPReductionPHIRecipe *clone() override {
@@ -2943,6 +2947,10 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
     return HasUsesOutsideReductionChain;
   }
 
+  void setExpressionSunk(bool V = true) { ExpressionSunk = V; }
+
+  bool isExpressionSunk() const { return ExpressionSunk; }
+
   /// Returns true if the recipe only uses the first lane of operand \p Op.
   bool usesFirstLaneOnly(const VPValue *Op) const override {
     assert(is_contained(operands(), Op) &&
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index dd85eeaa8592c..a75af19cab500 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -4573,10 +4573,13 @@ void VPlanTransforms::optimizeFindIVReductions(VPlan &Plan,
 
     // If IVOfExpressionToSink is an expression to sink, sink it now.
     VPValue *VectorRegionExitingVal = ReducedIV;
-    if (IVOfExpressionToSink)
+    bool SunkExpression = false;
+    if (IVOfExpressionToSink) {
       VectorRegionExitingVal =
           cloneBinOpForScalarIV(cast<VPWidenRecipe>(FindLastExpression),
                                 ReducedIV, IVOfExpressionToSink);
+      SunkExpression = true;
+    }
 
     VPValue *NewRdxResult;
     VPValue *StartVPV = PhiR->getStartValue();
@@ -4617,6 +4620,8 @@ void VPlanTransforms::optimizeFindIVReductions(VPlan &Plan,
         cast<PHINode>(PhiR->getUnderlyingInstr()), RecurKind::FindIV, *StartVPV,
         *NewFindLastSelect, RdxUnordered{1}, {},
         PhiR->hasUsesOutsideReductionChain());
+    if (SunkExpression)
+      NewPhiR->setExpressionSunk();
     NewPhiR->insertBefore(PhiR);
     PhiR->replaceAllUsesWith(NewPhiR);
     PhiR->eraseFromParent();
diff --git a/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll b/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
new file mode 100644
index 0000000000000..db46f9ee616d2
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
@@ -0,0 +1,52 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -force-vector-interleave=1 -force-vector-width=4 -enable-epilogue-vectorization -epilogue-vectorization-force-VF=4 -S %s | FileCheck %s
+
+; Test for https://github.com/llvm/llvm-project/issues/219211
+
+; CHECK-LABEL: define i32 @findiv_mul_pow2_sunk(
+; CHECK:       vector.body:
+; CHECK:       middle.block:
+; CHECK:       shl i32 {{.*}}, 2
+; CHECK-NOT:   vec.epilog
+; CHECK:       ret i32
+define i32 @findiv_mul_pow2_sunk(ptr %a, i32 %n) #0 {
+entry:
+  br label %loop
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+  %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+  %l = load i32, ptr %gep, align 4
+  %c = icmp eq i32 %l, 42
+  %expr = mul i32 %iv, 4
+  %sel = select i1 %c, i32 %expr, i32 %rdx
+  %iv.next = add nuw nsw i32 %iv, 1
+  %ec = icmp eq i32 %iv.next, %n
+  br i1 %ec, label %done, label %loop
+done:
+  ret i32 %sel
+}
+
+; CHECK-LABEL: define i32 @findiv_no_sunk_raw_iv(
+; CHECK:       vector.body:
+; CHECK:       middle.block:
+; CHECK-NOT:   shl i32
+; CHECK:       ret i32
+define i32 @findiv_no_sunk_raw_iv(ptr %a, i32 %n) #0 {
+entry:
+  br label %loop
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+  %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+  %l = load i32, ptr %gep, align 4
+  %c = icmp eq i32 %l, 42
+  %sel = select i1 %c, i32 %iv, i32 %rdx
+  %iv.next = add nuw nsw i32 %iv, 1
+  %ec = icmp eq i32 %iv.next, %n
+  br i1 %ec, label %done, label %loop
+done:
+  ret i32 %sel
+}
+
+attributes #0 = { "target-features"="+avx512f" }

>From 20b59cb7928a7695b0395222d52ee98722255a2e Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Wed, 2 Sep 2026 00:13:34 +0600
Subject: [PATCH 2/4] rewrite test and regenerate CHECK's

---
 .../Vectorize/LoopVectorizationPlanner.cpp    |   5 +
 llvm/lib/Transforms/Vectorize/VPlan.h         |   5 +-
 .../X86/find-iv-sunk-expr-epilogue.ll         |  52 -------
 .../find-iv-sunk-expr-epilogue.ll             | 131 ++++++++++++++++++
 4 files changed, 140 insertions(+), 53 deletions(-)
 delete mode 100644 llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
 create mode 100644 llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
index 82634ffb47b9d..0926781a02902 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorizationPlanner.cpp
@@ -879,6 +879,11 @@ static bool hasUnsupportedHeaderPhiRecipe(VPlan &Plan) {
               RecurrenceDescriptor::isFindLastRecurrenceKind(Kind) ||
               !RedPhi->getUnderlyingValue())
             return true;
+          // TODO: Add support for FindIV reductions with sunk expressions: the
+          // resume value from the main loop is in expression domain (e.g.,
+          // mul(ReducedIV, 3)), but the epilogue tracks raw IV values. A sunk
+          // expression is identified by a non-VPInstruction user of
+          // ComputeReductionResult.
           if (RecurrenceDescriptor::isFindIVRecurrenceKind(Kind))
             return RedPhi->isExpressionSunk();
           return false;
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index aaa456ee2e435..9d0fbb182a255 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -2879,6 +2879,9 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
   /// compare has multiple uses.
   bool HasUsesOutsideReductionChain;
 
+  // True if FindIV's expression was sunk into the vector loop. Epilogue
+  // is disabled while this is true. Temporary until epilogue handles sunk
+  // expressions.
   bool ExpressionSunk = false;
 
 public:
@@ -2947,7 +2950,7 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
     return HasUsesOutsideReductionChain;
   }
 
-  void setExpressionSunk(bool V = true) { ExpressionSunk = V; }
+  void setExpressionSunk() { ExpressionSunk = true; }
 
   bool isExpressionSunk() const { return ExpressionSunk; }
 
diff --git a/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll b/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
deleted file mode 100644
index db46f9ee616d2..0000000000000
--- a/llvm/test/Transforms/LoopVectorize/X86/find-iv-sunk-expr-epilogue.ll
+++ /dev/null
@@ -1,52 +0,0 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
-; RUN: opt -passes=loop-vectorize -force-vector-interleave=1 -force-vector-width=4 -enable-epilogue-vectorization -epilogue-vectorization-force-VF=4 -S %s | FileCheck %s
-
-; Test for https://github.com/llvm/llvm-project/issues/219211
-
-; CHECK-LABEL: define i32 @findiv_mul_pow2_sunk(
-; CHECK:       vector.body:
-; CHECK:       middle.block:
-; CHECK:       shl i32 {{.*}}, 2
-; CHECK-NOT:   vec.epilog
-; CHECK:       ret i32
-define i32 @findiv_mul_pow2_sunk(ptr %a, i32 %n) #0 {
-entry:
-  br label %loop
-loop:
-  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
-  %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
-  %gep = getelementptr inbounds i32, ptr %a, i32 %iv
-  %l = load i32, ptr %gep, align 4
-  %c = icmp eq i32 %l, 42
-  %expr = mul i32 %iv, 4
-  %sel = select i1 %c, i32 %expr, i32 %rdx
-  %iv.next = add nuw nsw i32 %iv, 1
-  %ec = icmp eq i32 %iv.next, %n
-  br i1 %ec, label %done, label %loop
-done:
-  ret i32 %sel
-}
-
-; CHECK-LABEL: define i32 @findiv_no_sunk_raw_iv(
-; CHECK:       vector.body:
-; CHECK:       middle.block:
-; CHECK-NOT:   shl i32
-; CHECK:       ret i32
-define i32 @findiv_no_sunk_raw_iv(ptr %a, i32 %n) #0 {
-entry:
-  br label %loop
-loop:
-  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
-  %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
-  %gep = getelementptr inbounds i32, ptr %a, i32 %iv
-  %l = load i32, ptr %gep, align 4
-  %c = icmp eq i32 %l, 42
-  %sel = select i1 %c, i32 %iv, i32 %rdx
-  %iv.next = add nuw nsw i32 %iv, 1
-  %ec = icmp eq i32 %iv.next, %n
-  br i1 %ec, label %done, label %loop
-done:
-  ret i32 %sel
-}
-
-attributes #0 = { "target-features"="+avx512f" }
diff --git a/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll b/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll
new file mode 100644
index 0000000000000..ba2a91f97e274
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll
@@ -0,0 +1,131 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
+; RUN: opt -passes=loop-vectorize -force-vector-interleave=1 -force-vector-width=4 -enable-epilogue-vectorization -epilogue-vectorization-force-VF=4 -S %s | FileCheck %s
+
+; Test for https://github.com/llvm/llvm-project/issues/219211
+
+define i32 @findiv_mul_pow2_sunk(ptr %a, i32 %n) {
+; CHECK-LABEL: define i32 @findiv_mul_pow2_sunk(
+; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i32 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ splat (i32 -2147483648), %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq <4 x i32> [[WIDE_LOAD]], splat (i32 42)
+; CHECK-NEXT:    [[TMP3]] = select <4 x i1> [[TMP2]], <4 x i32> [[VEC_IND]], <4 x i32> [[VEC_PHI]]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i32> [[VEC_IND]], splat (i32 4)
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP5:%.*]] = call i32 @llvm.vector.reduce.smax.v4i32(<4 x i32> [[TMP3]])
+; CHECK-NEXT:    [[TMP6:%.*]] = shl i32 [[TMP5]], 2
+; CHECK-NEXT:    [[TMP7:%.*]] = icmp ne i32 [[TMP5]], -2147483648
+; CHECK-NEXT:    [[TMP8:%.*]] = select i1 [[TMP7]], i32 [[TMP6]], i32 -1
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[DONE:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP8]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[RDX:%.*]] = phi i32 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[SEL:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[IV]]
+; CHECK-NEXT:    [[L:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[C:%.*]] = icmp eq i32 [[L]], 42
+; CHECK-NEXT:    [[EXPR:%.*]] = mul i32 [[IV]], 4
+; CHECK-NEXT:    [[SEL]] = select i1 [[C]], i32 [[EXPR]], i32 [[RDX]]
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i32 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[DONE]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+; CHECK:       [[DONE]]:
+; CHECK-NEXT:    [[SEL_LCSSA:%.*]] = phi i32 [ [[SEL]], %[[LOOP]] ], [ [[TMP8]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[SEL_LCSSA]]
+;
+entry:
+  br label %loop
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+  %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+  %l = load i32, ptr %gep, align 4
+  %c = icmp eq i32 %l, 42
+  %expr = mul i32 %iv, 4
+  %sel = select i1 %c, i32 %expr, i32 %rdx
+  %iv.next = add nuw nsw i32 %iv, 1
+  %ec = icmp eq i32 %iv.next, %n
+  br i1 %ec, label %done, label %loop
+done:
+  ret i32 %sel
+}
+
+define i32 @findiv_no_sunk_raw_iv(ptr %a, i32 %n) {
+; CHECK-LABEL: define i32 @findiv_no_sunk_raw_iv(
+; CHECK-SAME: ptr [[A:%.*]], i32 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[N]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = and i32 [[N]], 3
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[N]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i32> [ splat (i32 -2147483648), %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[TMP1:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[INDEX]]
+; CHECK-NEXT:    [[WIDE_LOAD:%.*]] = load <4 x i32>, ptr [[TMP1]], align 4
+; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq <4 x i32> [[WIDE_LOAD]], splat (i32 42)
+; CHECK-NEXT:    [[TMP3]] = select <4 x i1> [[TMP2]], <4 x i32> [[VEC_IND]], <4 x i32> [[VEC_PHI]]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT:    [[VEC_IND_NEXT]] = add nuw nsw <4 x i32> [[VEC_IND]], splat (i32 4)
+; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[TMP5:%.*]] = call i32 @llvm.vector.reduce.smax.v4i32(<4 x i32> [[TMP3]])
+; CHECK-NEXT:    [[TMP6:%.*]] = icmp ne i32 [[TMP5]], -2147483648
+; CHECK-NEXT:    [[TMP7:%.*]] = select i1 [[TMP6]], i32 [[TMP5]], i32 -1
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[N]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[DONE:.*]], label %[[SCALAR_PH]]
+; CHECK:       [[SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX:%.*]] = phi i32 [ [[TMP7]], %[[MIDDLE_BLOCK]] ], [ -1, %[[ENTRY]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[RDX:%.*]] = phi i32 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[SEL:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[GEP:%.*]] = getelementptr inbounds i32, ptr [[A]], i32 [[IV]]
+; CHECK-NEXT:    [[L:%.*]] = load i32, ptr [[GEP]], align 4
+; CHECK-NEXT:    [[C:%.*]] = icmp eq i32 [[L]], 42
+; CHECK-NEXT:    [[SEL]] = select i1 [[C]], i32 [[IV]], i32 [[RDX]]
+; CHECK-NEXT:    [[IV_NEXT]] = add nuw nsw i32 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp eq i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[DONE]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK:       [[DONE]]:
+; CHECK-NEXT:    [[SEL_LCSSA:%.*]] = phi i32 [ [[SEL]], %[[LOOP]] ], [ [[TMP7]], %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[SEL_LCSSA]]
+;
+entry:
+  br label %loop
+loop:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+  %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
+  %gep = getelementptr inbounds i32, ptr %a, i32 %iv
+  %l = load i32, ptr %gep, align 4
+  %c = icmp eq i32 %l, 42
+  %sel = select i1 %c, i32 %iv, i32 %rdx
+  %iv.next = add nuw nsw i32 %iv, 1
+  %ec = icmp eq i32 %iv.next, %n
+  br i1 %ec, label %done, label %loop
+done:
+  ret i32 %sel
+}

>From e9db3bde05a562babae2a7d8bebb2fac2fa5d4e2 Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Wed, 23 Sep 2026 15:07:30 +0600
Subject: [PATCH 3/4] use Doxygen format instead

---
 llvm/lib/Transforms/Vectorize/VPlan.h | 6 +++---
 1 file changed, 3 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index 9d0fbb182a255..fa9becbb4148d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -2879,9 +2879,9 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
   /// compare has multiple uses.
   bool HasUsesOutsideReductionChain;
 
-  // True if FindIV's expression was sunk into the vector loop. Epilogue
-  // is disabled while this is true. Temporary until epilogue handles sunk
-  // expressions.
+  /// True if FindIV's expression was sunk into the vector loop. Epilogue
+  /// is disabled while this is true. Temporary until epilogue handles sunk
+  /// expressions.
   bool ExpressionSunk = false;
 
 public:

>From 8b86e3348af564460de96e985ec75fcbe70359cb Mon Sep 17 00:00:00 2001
From: im-lunex <thisissamir04 at gmail.com>
Date: Thu, 1 Oct 2026 09:46:59 +0600
Subject: [PATCH 4/4] improve comment and address reviews and add todo

---
 llvm/lib/Transforms/Vectorize/VPlan.h                  | 10 +++++++---
 .../LoopVectorize/find-iv-sunk-expr-epilogue.ll        |  2 ++
 2 files changed, 9 insertions(+), 3 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index fa9becbb4148d..f9b539463a4a2 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -2879,9 +2879,13 @@ class VPReductionPHIRecipe : public VPHeaderPHIRecipe, public VPIRFlags {
   /// compare has multiple uses.
   bool HasUsesOutsideReductionChain;
 
-  /// True if FindIV's expression was sunk into the vector loop. Epilogue
-  /// is disabled while this is true. Temporary until epilogue handles sunk
-  /// expressions.
+  /// Temporary flag indicating that the FindIV reduction expression has been
+  /// sunk. While this is true, epilogue vectorization is
+  /// disabled to avoid applying the expression twice (once in the main vector
+  /// loop and again in the epilogue), which can produce incorrect results by
+  /// doing the sunk operation twice.
+  /// TODO: Remove this flag once epilogue vectorization properly supports sunk
+  /// FindIV expressions.
   bool ExpressionSunk = false;
 
 public:
diff --git a/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll b/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll
index ba2a91f97e274..d1398022508ec 100644
--- a/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll
+++ b/llvm/test/Transforms/LoopVectorize/find-iv-sunk-expr-epilogue.ll
@@ -116,6 +116,7 @@ define i32 @findiv_no_sunk_raw_iv(ptr %a, i32 %n) {
 ;
 entry:
   br label %loop
+
 loop:
   %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
   %rdx = phi i32 [ -1, %entry ], [ %sel, %loop ]
@@ -126,6 +127,7 @@ loop:
   %iv.next = add nuw nsw i32 %iv, 1
   %ec = icmp eq i32 %iv.next, %n
   br i1 %ec, label %done, label %loop
+
 done:
   ret i32 %sel
 }



More information about the llvm-commits mailing list