[llvm] [VPlan] - Generalize dead cycle removal in removeDeadRecipes. (PR #213632)
Pawan Nirpal via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 31 00:49:35 PDT 2026
https://github.com/pawan-nirpal-031 updated https://github.com/llvm/llvm-project/pull/213632
>From 1aa36de990cc38d16f07b0cc7c8a87033c59e980 Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Mon, 3 Aug 2026 02:25:55 -0700
Subject: [PATCH 1/5] [LV] - Fix crash in epilogue vectorization when AnyOf
reduction has no ComputeReductionResult
---
.../Transforms/Vectorize/LoopVectorize.cpp | 13 +++++---
...pilog-vectorization-anyof-no-rdx-result.ll | 31 +++++++++++++++++++
2 files changed, 40 insertions(+), 4 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index b45b8af5bf846..61642e63c2e5c 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7533,14 +7533,19 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
// TODO: Move setting of resume values to prepareToExecute.
if (auto *ReductionPhi = dyn_cast<VPReductionPHIRecipe>(&R)) {
// Find the reduction result by searching users of the phi or its backedge
- // value.
+ // value, looking through intermediate recipes.
auto IsReductionResult = [](VPRecipeBase *R) {
auto *VPI = dyn_cast<VPInstruction>(R);
return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
};
- auto *RdxResult = cast<VPInstruction>(
- vputils::findRecipe(ReductionPhi->getBackedgeValue(), IsReductionResult));
- assert(RdxResult && "expected to find reduction result");
+ auto *RdxResult = dyn_cast_or_null<VPInstruction>(vputils::findRecipe(
+ ReductionPhi->getBackedgeValue(), IsReductionResult));
+ // If the ComputeReductionResult was optimized away (e.g., the exit value
+ // was simplified to the start value), the reduction does not contribute
+ // to the exit value. Skip updating the start value for the epilogue,
+ // keeping it as the identity value.
+ if (!RdxResult)
+ continue;
VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
cast<PHINode>(ReductionPhi->getUnderlyingInstr()));
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
new file mode 100644
index 0000000000000..d7fea690e4692
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -0,0 +1,31 @@
+; RUN: opt -passes=loop-vectorize -S %s | FileCheck %s
+;
+; Verify that epilogue vectorization does not crash when a VPReductionPHIRecipe
+; (AnyOf reduction) has no ComputeReductionResult because the exit value was
+; simplified to a constant.
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+; CHECK-LABEL: @main
+; CHECK: vec.epilog.vector.body:
+; CHECK: middle.block:
+define i8 @main() {
+entry:
+ br label %loop
+
+exit:
+ ret i8 %sel
+
+loop:
+ %phi.rdx = phi i8 [ %sel, %loop ], [ 0, %entry ]
+ %iv = phi i32 [ %iv.next, %loop ], [ 1, %entry ]
+ %cmp = icmp sgt i32 0, 0
+ %sel = select i1 %cmp, i8 0, i8 %phi.rdx
+ %iv.next = add i32 %iv, 1
+ %exit.cond = icmp eq i32 %iv.next, 0
+ br i1 %exit.cond, label %exit, label %loop
+
+; uselistorder directives
+ uselistorder i8 %sel, { 1, 0 }
+}
>From fce017de87f68270620bb6110d0e5fc416e545dc Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pawannirpal at gmail.com>
Date: Mon, 3 Aug 2026 15:09:50 +0530
Subject: [PATCH 2/5] Update epilog-vectorization-anyof-no-rdx-result.ll
---
.../LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll | 3 ---
1 file changed, 3 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
index d7fea690e4692..efaeeca1169c8 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -1,8 +1,5 @@
; RUN: opt -passes=loop-vectorize -S %s | FileCheck %s
;
-; Verify that epilogue vectorization does not crash when a VPReductionPHIRecipe
-; (AnyOf reduction) has no ComputeReductionResult because the exit value was
-; simplified to a constant.
target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
target triple = "x86_64-unknown-linux-gnu"
>From edacb3c39011e33a3a864e84305b45ce3f2bfc11 Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Mon, 3 Aug 2026 02:25:55 -0700
Subject: [PATCH 3/5] [LV] - Fix crash in epilogue vectorization when AnyOf
reduction has no ComputeReductionResult
---
.../Transforms/Vectorize/LoopVectorize.cpp | 20 ++++--
...pilog-vectorization-anyof-no-rdx-result.ll | 70 ++++++++++++++++---
2 files changed, 73 insertions(+), 17 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 61642e63c2e5c..9f512f5d167c2 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7528,7 +7528,7 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
SmallVector<Instruction *> InstsToMove;
// Ensure that the start values for all header phi recipes are updated before
// vectorizing the epilogue loop.
- for (VPRecipeBase &R : Header->phis()) {
+ for (VPRecipeBase &R : make_early_inc_range(Header->phis())) {
Value *ResumeV = nullptr;
// TODO: Move setting of resume values to prepareToExecute.
if (auto *ReductionPhi = dyn_cast<VPReductionPHIRecipe>(&R)) {
@@ -7538,14 +7538,22 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
auto *VPI = dyn_cast<VPInstruction>(R);
return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
};
- auto *RdxResult = dyn_cast_or_null<VPInstruction>(vputils::findRecipe(
- ReductionPhi->getBackedgeValue(), IsReductionResult));
+ VPValue *BackedgeVal = ReductionPhi->getBackedgeValue();
+ auto *RdxResult = cast_or_null<VPInstruction>(
+ vputils::findRecipe(BackedgeVal, IsReductionResult));
// If the ComputeReductionResult was optimized away (e.g., the exit value
// was simplified to the start value), the reduction does not contribute
- // to the exit value. Skip updating the start value for the epilogue,
- // keeping it as the identity value.
- if (!RdxResult)
+ // to the exit value. Get rid of the dead reduction cycle.
+ if (!RdxResult) {
+ ReductionPhi->replaceAllUsesWith(ReductionPhi->getStartValue());
+ // The backedge value may be the phi itself (if the backedge chain was
+ // simplified away), or a separate recipe forming a cycle with the phi.
+ bool BackedgeIsPhi = (BackedgeVal == ReductionPhi);
+ ReductionPhi->eraseFromParent();
+ if (!BackedgeIsPhi)
+ vputils::recursivelyDeleteDeadRecipes(BackedgeVal);
continue;
+ }
VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
cast<PHINode>(ReductionPhi->getUnderlyingInstr()));
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
index efaeeca1169c8..f80dce8cbafc5 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -1,13 +1,56 @@
-; RUN: opt -passes=loop-vectorize -S %s | FileCheck %s
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -enable-epilogue-vectorization -epilogue-vectorization-force-VF=2 -S %s \
+; RUN: | FileCheck %s
;
+; Verify that epilogue vectorization does not crash when a VPReductionPHIRecipe
+; (AnyOf reduction) has no ComputeReductionResult because the exit value was
+; simplified to a constant. The dead reduction cycle should be deleted.
-target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
-target triple = "x86_64-unknown-linux-gnu"
-
-; CHECK-LABEL: @main
-; CHECK: vec.epilog.vector.body:
-; CHECK: middle.block:
-define i8 @main() {
+define i8 @dead_anyof_reduction() {
+; CHECK-LABEL: define i8 @dead_anyof_reduction() {
+; CHECK-NEXT: [[ITER_CHECK:.*]]:
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[VEC_PHI]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], -4
+; CHECK-NEXT: br i1 [[TMP0]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK: [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
+; CHECK: [[VEC_EPILOG_PH]]:
+; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ -4, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK: [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX1:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT2:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT2]] = add nuw i32 [[INDEX1]], 2
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq i32 [[INDEX_NEXT2]], -2
+; CHECK-NEXT: br i1 [[TMP1]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT: br i1 false, label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT: [[BC_MERGE_RDX3:%.*]] = phi i8 [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT: [[BC_RESUME_VAL4:%.*]] = phi i32 [ -1, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ -3, %[[VEC_EPILOG_ITER_CHECK]] ], [ 1, %[[ITER_CHECK]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[SEL_LCSSA:%.*]] = phi i8 [ [[SEL:%.*]], %[[LOOP]] ], [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i8 [[SEL_LCSSA]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[PHI_RDX:%.*]] = phi i8 [ [[SEL]], %[[LOOP]] ], [ [[BC_MERGE_RDX3]], %[[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[LOOP]] ], [ [[BC_RESUME_VAL4]], %[[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-NEXT: [[CMP:%.*]] = icmp sgt i32 0, 0
+; CHECK-NEXT: [[SEL]] = select i1 [[CMP]], i8 0, i8 [[PHI_RDX]]
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT: [[EXIT_COND:%.*]] = icmp eq i32 [[IV_NEXT]], 0
+; CHECK-NEXT: br i1 [[EXIT_COND]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+;
entry:
br label %loop
@@ -22,7 +65,12 @@ loop:
%iv.next = add i32 %iv, 1
%exit.cond = icmp eq i32 %iv.next, 0
br i1 %exit.cond, label %exit, label %loop
-
-; uselistorder directives
- uselistorder i8 %sel, { 1, 0 }
}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[PROF3]] = !{!"branch_weights", i32 2, i32 2}
+; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META1]], [[META2]]}
+; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META2]], [[META1]]}
+;.
>From a533c60eaebfe57a0784343ba28bd4f962ab693d Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Mon, 3 Aug 2026 02:25:55 -0700
Subject: [PATCH 4/5] [LV] - Fix crash in epilogue vectorization when AnyOf
reduction has no ComputeReductionResult
---
.../Transforms/Vectorize/LoopVectorize.cpp | 23 +----
.../Transforms/Vectorize/VPlanTransforms.cpp | 43 ++++++---
...pilog-vectorization-anyof-no-rdx-result.ll | 93 +++++++++++++++++--
.../LoopVectorize/select-cmp-blend-chain.ll | 33 -------
4 files changed, 117 insertions(+), 75 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 9f512f5d167c2..b45b8af5bf846 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7528,32 +7528,19 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
SmallVector<Instruction *> InstsToMove;
// Ensure that the start values for all header phi recipes are updated before
// vectorizing the epilogue loop.
- for (VPRecipeBase &R : make_early_inc_range(Header->phis())) {
+ for (VPRecipeBase &R : Header->phis()) {
Value *ResumeV = nullptr;
// TODO: Move setting of resume values to prepareToExecute.
if (auto *ReductionPhi = dyn_cast<VPReductionPHIRecipe>(&R)) {
// Find the reduction result by searching users of the phi or its backedge
- // value, looking through intermediate recipes.
+ // value.
auto IsReductionResult = [](VPRecipeBase *R) {
auto *VPI = dyn_cast<VPInstruction>(R);
return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
};
- VPValue *BackedgeVal = ReductionPhi->getBackedgeValue();
- auto *RdxResult = cast_or_null<VPInstruction>(
- vputils::findRecipe(BackedgeVal, IsReductionResult));
- // If the ComputeReductionResult was optimized away (e.g., the exit value
- // was simplified to the start value), the reduction does not contribute
- // to the exit value. Get rid of the dead reduction cycle.
- if (!RdxResult) {
- ReductionPhi->replaceAllUsesWith(ReductionPhi->getStartValue());
- // The backedge value may be the phi itself (if the backedge chain was
- // simplified away), or a separate recipe forming a cycle with the phi.
- bool BackedgeIsPhi = (BackedgeVal == ReductionPhi);
- ReductionPhi->eraseFromParent();
- if (!BackedgeIsPhi)
- vputils::recursivelyDeleteDeadRecipes(BackedgeVal);
- continue;
- }
+ auto *RdxResult = cast<VPInstruction>(
+ vputils::findRecipe(ReductionPhi->getBackedgeValue(), IsReductionResult));
+ assert(RdxResult && "expected to find reduction result");
VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
cast<PHINode>(ReductionPhi->getUnderlyingInstr()));
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 9027846fa54f9..340b61712268c 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -687,6 +687,32 @@ static void removeRedundantInductionCasts(VPlan &Plan) {
}
}
+/// If R is a phi-like recipe starting a dead cycle of recipes, erase all
+/// reachable recipes of the dead cycle.
+static void tryToRemoveDeadCycle(VPRecipeBase *R) {
+ auto *PhiR = dyn_cast<VPSingleDefRecipe>(R);
+ if (!PhiR || !isa<VPPhiAccessors>(R) || isa<VPCurrentIterationPHIRecipe>(R))
+ return;
+
+ // The transitive users of PhiR are closed under users, so the cycle is dead
+ // if every one of them can be erased.
+ for (VPUser *U : vputils::collectUsersRecursively(PhiR)) {
+ auto *R = cast<VPRecipeBase>(U);
+ // Bail out if a user must be retained, or if it is a phi-like recipe other
+ // than PhiR;
+ if (R->mayHaveSideEffects() || (R != PhiR && isa<VPPhiAccessors>(R)))
+ return;
+ }
+
+ // Break the cycle by replacing PhiR with its first incoming value, which is
+ // defined outside the cycle. That leaves the rest of the cycle dead.
+ PhiR->replaceAllUsesWith(PhiR->getOperand(0));
+ SmallVector<VPValue *> Incoming(PhiR->operands());
+ PhiR->eraseFromParent();
+ for (VPValue *Op : Incoming)
+ vputils::recursivelyDeleteDeadRecipes(Op);
+}
+
void VPlanTransforms::removeDeadRecipes(VPlan &Plan) {
PostOrderTraversal<VPBlockDeepTraversalWrapper<VPBlockBase *>> POT(
Plan.getEntry());
@@ -699,20 +725,9 @@ void VPlanTransforms::removeDeadRecipes(VPlan &Plan) {
continue;
}
- // Check if R is a dead VPPhi <-> update cycle and remove it.
- VPValue *Start, *Incoming;
- if (!match(&R, m_VPPhi(m_VPValue(Start), m_VPValue(Incoming))))
- continue;
- auto *PhiR = cast<VPPhi>(&R);
- VPUser *PhiUser = PhiR->getSingleUser();
- if (!PhiUser)
- continue;
- if (PhiUser != Incoming->getDefiningRecipe() ||
- Incoming->getNumUsers() != 1)
- continue;
- PhiR->replaceAllUsesWith(Start);
- PhiR->eraseFromParent();
- Incoming->getDefiningRecipe()->eraseFromParent();
+ // If R is a phi-like recipe starting a dead cycle of recipes, erase the
+ // whole cycle.
+ tryToRemoveDeadCycle(&R);
}
}
}
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
index f80dce8cbafc5..8c74af2508c9f 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
; RUN: -enable-epilogue-vectorization -epilogue-vectorization-force-VF=2 -S %s \
; RUN: | FileCheck %s
@@ -17,7 +17,6 @@ define i8 @dead_anyof_reduction() {
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[VEC_PHI]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
; CHECK-NEXT: [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], -4
; CHECK-NEXT: br i1 [[TMP0]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
@@ -66,11 +65,85 @@ loop:
%exit.cond = icmp eq i32 %iv.next, 0
br i1 %exit.cond, label %exit, label %loop
}
-;.
-; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
-; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
-; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
-; CHECK: [[PROF3]] = !{!"branch_weights", i32 2, i32 2}
-; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META1]], [[META2]]}
-; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META2]], [[META1]]}
-;.
+
+define i32 @dead_anyof_reduction_blend(i1 %c0, i32 %n) {
+; CHECK-LABEL: define i32 @dead_anyof_reduction_blend(
+; CHECK-SAME: i1 [[C0:%.*]], i32 [[N:%.*]]) {
+; CHECK-NEXT: [[ITER_CHECK:.*]]:
+; CHECK-NEXT: [[SMAX:%.*]] = call i32 @llvm.smax.i32(i32 [[N]], i32 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[SMAX]], 2
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i32 [[SMAX]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP0:%.*]] = urem i32 [[SMAX]], 4
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[SMAX]], [[TMP0]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP1]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[SMAX]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK: [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP0]], 2
+; CHECK-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; CHECK: [[VEC_EPILOG_PH]]:
+; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = urem i32 [[SMAX]], 2
+; CHECK-NEXT: [[N_VEC2:%.*]] = sub i32 [[SMAX]], [[TMP2]]
+; CHECK-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK: [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX3:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT4]] = add nuw i32 [[INDEX3]], 2
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT4]], [[N_VEC2]]
+; CHECK-NEXT: br i1 [[TMP3]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N5:%.*]] = icmp eq i32 [[SMAX]], [[N_VEC2]]
+; CHECK-NEXT: br i1 [[CMP_N5]], label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC2]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT: [[BC_MERGE_RDX7:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[RDX:%.*]] = phi i32 [ [[BC_MERGE_RDX7]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[RDX_NEXT:%.*]], %[[LOOP_LATCH]] ]
+; CHECK-NEXT: br i1 [[C0]], label %[[LOOP_LATCH]], label %[[IF_THEN:.*]]
+; CHECK: [[IF_THEN]]:
+; CHECK-NEXT: [[C:%.*]] = icmp eq i32 [[IV]], 0
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[C]], i32 0, i32 [[RDX]]
+; CHECK-NEXT: br label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[RDX_NEXT]] = phi i32 [ [[SEL]], %[[IF_THEN]] ], [ [[RDX]], %[[LOOP_HEADER]] ]
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT: [[EC:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i32 [[RDX_NEXT_LCSSA]]
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %rdx = phi i32 [ 0, %entry ], [ %rdx.next, %loop.latch ]
+ br i1 %c0, label %loop.latch, label %if.then
+
+if.then:
+ %c = icmp eq i32 %iv, 0
+ %sel = select i1 %c, i32 0, i32 %rdx
+ br label %loop.latch
+
+loop.latch:
+ %rdx.next = phi i32 [ %sel, %if.then ], [ %rdx, %loop.header ]
+ %iv.next = add i32 %iv, 1
+ %ec = icmp slt i32 %iv.next, %n
+ br i1 %ec, label %loop.header, label %exit
+
+exit:
+ ret i32 %rdx.next
+}
diff --git a/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll b/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
index ada3dc5a22f95..d45c4cc93bdc7 100644
--- a/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
+++ b/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
@@ -15,14 +15,7 @@ define i32 @anyof_two_blend_chain(i1 %c0, i1 %c1, i32 %n) {
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI1:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT: [[TMP1:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP0]]
-; CHECK-NEXT: [[PREDPHI:%.*]] = select i1 [[C1]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP1]]
-; CHECK-NEXT: [[PREDPHI1]] = select i1 [[C0]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -97,15 +90,7 @@ define i32 @anyof_three_blend_chain(i1 %c0, i1 %c1, i1 %c2, i32 %n) {
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI2:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT: [[TMP1:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP0]]
-; CHECK-NEXT: [[PREDPHI:%.*]] = select i1 [[C2]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP1]]
-; CHECK-NEXT: [[PREDPHI1:%.*]] = select i1 [[C1]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI]]
-; CHECK-NEXT: [[PREDPHI2]] = select i1 [[C0]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI1]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -189,21 +174,10 @@ define i32 @anyof_diamond_blend_chain(i1 %c0, i1 %c1, i32 %n) {
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i32 [[SMAX]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[SMAX]], [[N_MOD_VF]]
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i1> poison, i1 [[C1]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT1]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <4 x i1> poison, i1 [[C0]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT3:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT2]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP1:%.*]] = select <4 x i1> [[BROADCAST_SPLAT2]], <4 x i1> [[BROADCAST_SPLAT3]], <4 x i1> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP0]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[TMP3]] = or <4 x i1> [[VEC_PHI]], [[TMP2]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -293,17 +267,10 @@ define i32 @anyof_blend_with_select_mask(i1 %a, i1 %b, i32 %n) {
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i32 [[SMAX]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[SMAX]], [[N_MOD_VF]]
-; CHECK-NEXT: [[TMP0:%.*]] = select i1 [[A]], i1 [[B]], i1 false
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP1]]
-; CHECK-NEXT: [[PREDPHI]] = select i1 [[TMP0]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP2]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
>From 6ebe5eef76f69804064340090def7afcafba39c9 Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Mon, 3 Aug 2026 02:25:55 -0700
Subject: [PATCH 5/5] [LV] - Fix crash in epilogue vectorization when AnyOf
reduction has no ComputeReductionResult
---
.../Transforms/Vectorize/VPlanTransforms.cpp | 2 +-
...pilog-vectorization-anyof-no-rdx-result.ll | 175 +++++++++++++++++-
.../LoopVectorize/select-cmp-blend-chain.ll | 77 ++++++--
3 files changed, 229 insertions(+), 25 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 340b61712268c..7944597f76309 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -691,7 +691,7 @@ static void removeRedundantInductionCasts(VPlan &Plan) {
/// reachable recipes of the dead cycle.
static void tryToRemoveDeadCycle(VPRecipeBase *R) {
auto *PhiR = dyn_cast<VPSingleDefRecipe>(R);
- if (!PhiR || !isa<VPPhiAccessors>(R) || isa<VPCurrentIterationPHIRecipe>(R))
+ if (!PhiR || !isa<VPPhi, VPReductionPHIRecipe>(R))
return;
// The transitive users of PhiR are closed under users, so the cycle is dead
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
index 8c74af2508c9f..593baf4f102ea 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -2,10 +2,9 @@
; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
; RUN: -enable-epilogue-vectorization -epilogue-vectorization-force-VF=2 -S %s \
; RUN: | FileCheck %s
-;
-; Verify that epilogue vectorization does not crash when a VPReductionPHIRecipe
-; (AnyOf reduction) has no ComputeReductionResult because the exit value was
-; simplified to a constant. The dead reduction cycle should be deleted.
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -force-vector-interleave=2 -S %s \
+; RUN: | FileCheck %s --check-prefix=CHECK-IL2
define i8 @dead_anyof_reduction() {
; CHECK-LABEL: define i8 @dead_anyof_reduction() {
@@ -50,6 +49,32 @@ define i8 @dead_anyof_reduction() {
; CHECK-NEXT: [[EXIT_COND:%.*]] = icmp eq i32 [[IV_NEXT]], 0
; CHECK-NEXT: br i1 [[EXIT_COND]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
;
+; CHECK-IL2-LABEL: define i8 @dead_anyof_reduction() {
+; CHECK-IL2-NEXT: [[ENTRY:.*:]]
+; CHECK-IL2-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK-IL2: [[VECTOR_PH]]:
+; CHECK-IL2-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK-IL2: [[VECTOR_BODY]]:
+; CHECK-IL2-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-IL2-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
+; CHECK-IL2-NEXT: [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], -8
+; CHECK-IL2-NEXT: br i1 [[TMP0]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK-IL2: [[MIDDLE_BLOCK]]:
+; CHECK-IL2-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK-IL2: [[SCALAR_PH]]:
+; CHECK-IL2-NEXT: br label %[[LOOP:.*]]
+; CHECK-IL2: [[EXIT:.*]]:
+; CHECK-IL2-NEXT: [[SEL_LCSSA:%.*]] = phi i8 [ [[SEL:%.*]], %[[LOOP]] ]
+; CHECK-IL2-NEXT: ret i8 [[SEL_LCSSA]]
+; CHECK-IL2: [[LOOP]]:
+; CHECK-IL2-NEXT: [[PHI_RDX:%.*]] = phi i8 [ [[SEL]], %[[LOOP]] ], [ 0, %[[SCALAR_PH]] ]
+; CHECK-IL2-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[LOOP]] ], [ -7, %[[SCALAR_PH]] ]
+; CHECK-IL2-NEXT: [[CMP:%.*]] = icmp sgt i32 0, 0
+; CHECK-IL2-NEXT: [[SEL]] = select i1 [[CMP]], i8 0, i8 [[PHI_RDX]]
+; CHECK-IL2-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-IL2-NEXT: [[EXIT_COND:%.*]] = icmp eq i32 [[IV_NEXT]], 0
+; CHECK-IL2-NEXT: br i1 [[EXIT_COND]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP3:![0-9]+]]
+;
entry:
br label %loop
@@ -77,7 +102,7 @@ define i32 @dead_anyof_reduction_blend(i1 %c0, i32 %n) {
; CHECK-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i32 [[SMAX]], 4
; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
; CHECK: [[VECTOR_PH]]:
-; CHECK-NEXT: [[TMP0:%.*]] = urem i32 [[SMAX]], 4
+; CHECK-NEXT: [[TMP0:%.*]] = and i32 [[SMAX]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[SMAX]], [[TMP0]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
@@ -93,7 +118,7 @@ define i32 @dead_anyof_reduction_blend(i1 %c0, i32 %n) {
; CHECK-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
; CHECK: [[VEC_EPILOG_PH]]:
; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
-; CHECK-NEXT: [[TMP2:%.*]] = urem i32 [[SMAX]], 2
+; CHECK-NEXT: [[TMP2:%.*]] = and i32 [[SMAX]], 1
; CHECK-NEXT: [[N_VEC2:%.*]] = sub i32 [[SMAX]], [[TMP2]]
; CHECK-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
; CHECK: [[VEC_EPILOG_VECTOR_BODY]]:
@@ -106,11 +131,11 @@ define i32 @dead_anyof_reduction_blend(i1 %c0, i32 %n) {
; CHECK-NEXT: br i1 [[CMP_N5]], label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC2]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
-; CHECK-NEXT: [[BC_MERGE_RDX7:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT: [[BC_MERGE_RDX6:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
; CHECK: [[LOOP_HEADER]]:
; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
-; CHECK-NEXT: [[RDX:%.*]] = phi i32 [ [[BC_MERGE_RDX7]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[RDX_NEXT:%.*]], %[[LOOP_LATCH]] ]
+; CHECK-NEXT: [[RDX:%.*]] = phi i32 [ [[BC_MERGE_RDX6]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[RDX_NEXT:%.*]], %[[LOOP_LATCH]] ]
; CHECK-NEXT: br i1 [[C0]], label %[[LOOP_LATCH]], label %[[IF_THEN:.*]]
; CHECK: [[IF_THEN]]:
; CHECK-NEXT: [[C:%.*]] = icmp eq i32 [[IV]], 0
@@ -125,6 +150,45 @@ define i32 @dead_anyof_reduction_blend(i1 %c0, i32 %n) {
; CHECK-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
; CHECK-NEXT: ret i32 [[RDX_NEXT_LCSSA]]
;
+; CHECK-IL2-LABEL: define i32 @dead_anyof_reduction_blend(
+; CHECK-IL2-SAME: i1 [[C0:%.*]], i32 [[N:%.*]]) {
+; CHECK-IL2-NEXT: [[ENTRY:.*]]:
+; CHECK-IL2-NEXT: [[TMP0:%.*]] = call i32 @llvm.smax.i32(i32 [[N]], i32 1)
+; CHECK-IL2-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP0]], 8
+; CHECK-IL2-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK-IL2: [[VECTOR_PH]]:
+; CHECK-IL2-NEXT: [[TMP1:%.*]] = and i32 [[TMP0]], 7
+; CHECK-IL2-NEXT: [[N_VEC:%.*]] = sub i32 [[TMP0]], [[TMP1]]
+; CHECK-IL2-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK-IL2: [[VECTOR_BODY]]:
+; CHECK-IL2-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-IL2-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
+; CHECK-IL2-NEXT: [[TMP2:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-IL2-NEXT: br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK-IL2: [[MIDDLE_BLOCK]]:
+; CHECK-IL2-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[TMP0]], [[N_VEC]]
+; CHECK-IL2-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
+; CHECK-IL2: [[SCALAR_PH]]:
+; CHECK-IL2-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-IL2-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-IL2-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK-IL2: [[LOOP_HEADER]]:
+; CHECK-IL2-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-IL2-NEXT: [[RDX:%.*]] = phi i32 [ [[BC_MERGE_RDX]], %[[SCALAR_PH]] ], [ [[RDX_NEXT:%.*]], %[[LOOP_LATCH]] ]
+; CHECK-IL2-NEXT: br i1 [[C0]], label %[[LOOP_LATCH]], label %[[IF_THEN:.*]]
+; CHECK-IL2: [[IF_THEN]]:
+; CHECK-IL2-NEXT: [[C:%.*]] = icmp eq i32 [[IV]], 0
+; CHECK-IL2-NEXT: [[SEL:%.*]] = select i1 [[C]], i32 0, i32 [[RDX]]
+; CHECK-IL2-NEXT: br label %[[LOOP_LATCH]]
+; CHECK-IL2: [[LOOP_LATCH]]:
+; CHECK-IL2-NEXT: [[RDX_NEXT]] = phi i32 [ [[SEL]], %[[IF_THEN]] ], [ [[RDX]], %[[LOOP_HEADER]] ]
+; CHECK-IL2-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-IL2-NEXT: [[EC:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
+; CHECK-IL2-NEXT: br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK-IL2: [[EXIT]]:
+; CHECK-IL2-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ 0, %[[MIDDLE_BLOCK]] ]
+; CHECK-IL2-NEXT: ret i32 [[RDX_NEXT_LCSSA]]
+;
entry:
br label %loop.header
@@ -147,3 +211,98 @@ loop.latch:
exit:
ret i32 %rdx.next
}
+
+define i8 @dead_truncated_induction() {
+; CHECK-LABEL: define i8 @dead_truncated_induction() {
+; CHECK-NEXT: [[ITER_CHECK:.*]]:
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i8> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[VEC_IND]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], 100
+; CHECK-NEXT: br i1 [[TMP0]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP9:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP1:%.*]] = extractelement <4 x i8> [[VEC_IND]], i64 3
+; CHECK-NEXT: br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK: [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; CHECK: [[VEC_EPILOG_PH]]:
+; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ 100, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ 25600, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = trunc i32 [[BC_RESUME_VAL]] to i8
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x i8> poison, i8 [[TMP2]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <2 x i8> [[BROADCAST_SPLATINSERT]], <2 x i8> poison, <2 x i32> zeroinitializer
+; CHECK-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK: [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX1:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT3:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND2:%.*]] = phi <2 x i8> [ [[BROADCAST_SPLAT]], %[[VEC_EPILOG_PH]] ], [ [[VEC_IND2]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT3]] = add nuw i32 [[INDEX1]], 2
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT3]], 102
+; CHECK-NEXT: br i1 [[TMP3]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP10:![0-9]+]]
+; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP4:%.*]] = extractelement <2 x i8> [[VEC_IND2]], i64 1
+; CHECK-NEXT: br i1 true, label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL4:%.*]] = phi i32 [ 102, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 100, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT: [[BC_RESUME_VAL5:%.*]] = phi i32 [ 26112, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 25600, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL4]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[IV_2:%.*]] = phi i32 [ [[BC_RESUME_VAL5]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[ADD:%.*]], %[[LOOP]] ]
+; CHECK-NEXT: [[ADD]] = add i32 [[IV_2]], 256
+; CHECK-NEXT: [[TRUNC:%.*]] = trunc i32 [[IV_2]] to i8
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT: [[EC:%.*]] = icmp ugt i32 [[IV]], 100
+; CHECK-NEXT: br i1 [[EC]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP11:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[RES:%.*]] = phi i8 [ [[TRUNC]], %[[LOOP]] ], [ [[TMP1]], %[[MIDDLE_BLOCK]] ], [ [[TMP4]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i8 [[RES]]
+;
+; CHECK-IL2-LABEL: define i8 @dead_truncated_induction() {
+; CHECK-IL2-NEXT: [[ENTRY:.*:]]
+; CHECK-IL2-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK-IL2: [[VECTOR_PH]]:
+; CHECK-IL2-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK-IL2: [[VECTOR_BODY]]:
+; CHECK-IL2-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-IL2-NEXT: [[VEC_IND:%.*]] = phi <4 x i8> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[VEC_IND]], %[[VECTOR_BODY]] ]
+; CHECK-IL2-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 8
+; CHECK-IL2-NEXT: [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], 96
+; CHECK-IL2-NEXT: br i1 [[TMP0]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK-IL2: [[MIDDLE_BLOCK]]:
+; CHECK-IL2-NEXT: br label %[[SCALAR_PH:.*]]
+; CHECK-IL2: [[SCALAR_PH]]:
+; CHECK-IL2-NEXT: br label %[[LOOP:.*]]
+; CHECK-IL2: [[LOOP]]:
+; CHECK-IL2-NEXT: [[IV:%.*]] = phi i32 [ 96, %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-IL2-NEXT: [[IV_2:%.*]] = phi i32 [ 24576, %[[SCALAR_PH]] ], [ [[ADD:%.*]], %[[LOOP]] ]
+; CHECK-IL2-NEXT: [[ADD]] = add i32 [[IV_2]], 256
+; CHECK-IL2-NEXT: [[TRUNC:%.*]] = trunc i32 [[IV_2]] to i8
+; CHECK-IL2-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-IL2-NEXT: [[EC:%.*]] = icmp ugt i32 [[IV]], 100
+; CHECK-IL2-NEXT: br i1 [[EC]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK-IL2: [[EXIT]]:
+; CHECK-IL2-NEXT: [[RES:%.*]] = phi i8 [ [[TRUNC]], %[[LOOP]] ]
+; CHECK-IL2-NEXT: ret i8 [[RES]]
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop ]
+ %iv.2 = phi i32 [ 0, %entry ], [ %add, %loop ]
+ %add = add i32 %iv.2, 256
+ %trunc = trunc i32 %iv.2 to i8
+ %iv.next = add i32 %iv, 1
+ %ec = icmp ugt i32 %iv, 100
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ %res = phi i8 [ %trunc, %loop ]
+ ret i8 %res
+}
diff --git a/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll b/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
index d45c4cc93bdc7..6cae52e2a8f28 100644
--- a/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
+++ b/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
@@ -15,15 +15,25 @@ define i32 @anyof_two_blend_chain(i1 %c0, i1 %c1, i32 %n) {
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI1:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
+; CHECK-NEXT: [[TMP3:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP4]]
+; CHECK-NEXT: [[PREDPHI:%.*]] = select i1 [[C1]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP3]]
+; CHECK-NEXT: [[PREDPHI1]] = select i1 [[C0]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[PREDPHI1]])
+; CHECK-NEXT: [[TMP6:%.*]] = freeze i1 [[TMP5]]
+; CHECK-NEXT: [[RDX_SELECT:%.*]] = select i1 [[TMP6]], i32 1, i32 0
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[SMAX]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
; CHECK: [[LOOP_HEADER]]:
; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
@@ -33,7 +43,7 @@ define i32 @anyof_two_blend_chain(i1 %c0, i1 %c1, i32 %n) {
; CHECK-NEXT: br i1 [[C1]], label %[[IF_MERGE:.*]], label %[[IF_INNER:.*]]
; CHECK: [[IF_INNER]]:
; CHECK-NEXT: [[C:%.*]] = icmp eq i32 [[IV]], 0
-; CHECK-NEXT: [[SEL:%.*]] = select i1 [[C]], i32 0, i32 [[RDX]]
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[C]], i32 1, i32 [[RDX]]
; CHECK-NEXT: br label %[[IF_MERGE]]
; CHECK: [[IF_MERGE]]:
; CHECK-NEXT: [[BLEND1:%.*]] = phi i32 [ [[SEL]], %[[IF_INNER]] ], [ [[RDX]], %[[IF_OUTER]] ]
@@ -44,7 +54,7 @@ define i32 @anyof_two_blend_chain(i1 %c0, i1 %c1, i32 %n) {
; CHECK-NEXT: [[EC:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
; CHECK-NEXT: br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP3:![0-9]+]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ 0, %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
; CHECK-NEXT: ret i32 [[RDX_NEXT_LCSSA]]
;
entry:
@@ -60,7 +70,7 @@ if.outer:
if.inner:
%c = icmp eq i32 %iv, 0
- %sel = select i1 %c, i32 0, i32 %rdx
+ %sel = select i1 %c, i32 1, i32 %rdx
br label %if.merge
if.merge:
@@ -90,15 +100,26 @@ define i32 @anyof_three_blend_chain(i1 %c0, i1 %c1, i1 %c2, i32 %n) {
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI2:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
+; CHECK-NEXT: [[TMP3:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP4]]
+; CHECK-NEXT: [[PREDPHI:%.*]] = select i1 [[C2]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP3]]
+; CHECK-NEXT: [[PREDPHI1:%.*]] = select i1 [[C1]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI]]
+; CHECK-NEXT: [[PREDPHI2]] = select i1 [[C0]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI1]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[PREDPHI2]])
+; CHECK-NEXT: [[TMP6:%.*]] = freeze i1 [[TMP5]]
+; CHECK-NEXT: [[RDX_SELECT:%.*]] = select i1 [[TMP6]], i32 1, i32 0
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[SMAX]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
; CHECK: [[LOOP_HEADER]]:
; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
@@ -110,7 +131,7 @@ define i32 @anyof_three_blend_chain(i1 %c0, i1 %c1, i1 %c2, i32 %n) {
; CHECK-NEXT: br i1 [[C2]], label %[[IF2_MERGE:.*]], label %[[IF3:.*]]
; CHECK: [[IF3]]:
; CHECK-NEXT: [[C:%.*]] = icmp eq i32 [[IV]], 0
-; CHECK-NEXT: [[SEL:%.*]] = select i1 [[C]], i32 0, i32 [[RDX]]
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[C]], i32 1, i32 [[RDX]]
; CHECK-NEXT: br label %[[IF2_MERGE]]
; CHECK: [[IF2_MERGE]]:
; CHECK-NEXT: [[BLEND1:%.*]] = phi i32 [ [[SEL]], %[[IF3]] ], [ [[RDX]], %[[IF2]] ]
@@ -124,7 +145,7 @@ define i32 @anyof_three_blend_chain(i1 %c0, i1 %c1, i1 %c2, i32 %n) {
; CHECK-NEXT: [[EC:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
; CHECK-NEXT: br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP5:![0-9]+]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ 0, %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
; CHECK-NEXT: ret i32 [[RDX_NEXT_LCSSA]]
;
entry:
@@ -143,7 +164,7 @@ if2:
if3:
%c = icmp eq i32 %iv, 0
- %sel = select i1 %c, i32 0, i32 %rdx
+ %sel = select i1 %c, i32 1, i32 %rdx
br label %if2.merge
if2.merge:
@@ -174,18 +195,32 @@ define i32 @anyof_diamond_blend_chain(i1 %c0, i1 %c1, i32 %n) {
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i32 [[SMAX]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[SMAX]], [[N_MOD_VF]]
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i1> poison, i1 [[C1]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i1> poison, i1 [[C0]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT1]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP2:%.*]] = select <4 x i1> [[BROADCAST_SPLAT]], <4 x i1> [[BROADCAST_SPLAT2]], <4 x i1> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP5:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
+; CHECK-NEXT: [[TMP6:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP3]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP5]] = or <4 x i1> [[VEC_PHI]], [[TMP6]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP7:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP5]])
+; CHECK-NEXT: [[TMP8:%.*]] = freeze i1 [[TMP7]]
+; CHECK-NEXT: [[RDX_SELECT:%.*]] = select i1 [[TMP8]], i32 1, i32 0
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[SMAX]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
; CHECK: [[LOOP_HEADER]]:
; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
@@ -193,7 +228,7 @@ define i32 @anyof_diamond_blend_chain(i1 %c0, i1 %c1, i32 %n) {
; CHECK-NEXT: br i1 [[C0]], label %[[IF_TRUE:.*]], label %[[IF_FALSE:.*]]
; CHECK: [[IF_TRUE]]:
; CHECK-NEXT: [[C:%.*]] = icmp eq i32 [[IV]], 0
-; CHECK-NEXT: [[SEL:%.*]] = select i1 [[C]], i32 0, i32 [[RDX]]
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[C]], i32 1, i32 [[RDX]]
; CHECK-NEXT: br label %[[MERGE:.*]]
; CHECK: [[IF_FALSE]]:
; CHECK-NEXT: br label %[[MERGE]]
@@ -212,7 +247,7 @@ define i32 @anyof_diamond_blend_chain(i1 %c0, i1 %c1, i32 %n) {
; CHECK-NEXT: [[EC:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
; CHECK-NEXT: br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP7:![0-9]+]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ 0, %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
; CHECK-NEXT: ret i32 [[RDX_NEXT_LCSSA]]
;
entry:
@@ -225,7 +260,7 @@ loop.header:
if.true:
%c = icmp eq i32 %iv, 0
- %sel = select i1 %c, i32 0, i32 %rdx
+ %sel = select i1 %c, i32 1, i32 %rdx
br label %merge
if.false:
@@ -267,18 +302,28 @@ define i32 @anyof_blend_with_select_mask(i1 %a, i1 %b, i32 %n) {
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[N_MOD_VF:%.*]] = and i32 [[SMAX]], 3
; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[SMAX]], [[N_MOD_VF]]
+; CHECK-NEXT: [[TMP2:%.*]] = select i1 [[A]], i1 [[B]], i1 false
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
+; CHECK-NEXT: [[TMP4:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP5]]
+; CHECK-NEXT: [[PREDPHI]] = select i1 [[TMP2]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP4]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[TMP6:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[PREDPHI]])
+; CHECK-NEXT: [[TMP7:%.*]] = freeze i1 [[TMP6]]
+; CHECK-NEXT: [[RDX_SELECT:%.*]] = select i1 [[TMP7]], i32 1, i32 0
; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[SMAX]], [[N_VEC]]
; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[SCALAR_PH]]
; CHECK: [[SCALAR_PH]]:
; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
-; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
+; CHECK-NEXT: [[BC_MERGE_RDX:%.*]] = phi i32 [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ], [ 0, %[[ENTRY]] ]
; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
; CHECK: [[LOOP_HEADER]]:
; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
@@ -287,7 +332,7 @@ define i32 @anyof_blend_with_select_mask(i1 %a, i1 %b, i32 %n) {
; CHECK-NEXT: br i1 [[COND]], label %[[LOOP_LATCH]], label %[[IF_THEN:.*]]
; CHECK: [[IF_THEN]]:
; CHECK-NEXT: [[CMP_INNER:%.*]] = icmp eq i32 [[IV]], 0
-; CHECK-NEXT: [[SEL:%.*]] = select i1 [[CMP_INNER]], i32 0, i32 [[RDX]]
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[CMP_INNER]], i32 1, i32 [[RDX]]
; CHECK-NEXT: br label %[[LOOP_LATCH]]
; CHECK: [[LOOP_LATCH]]:
; CHECK-NEXT: [[RDX_NEXT]] = phi i32 [ [[RDX]], %[[LOOP_HEADER]] ], [ [[SEL]], %[[IF_THEN]] ]
@@ -295,7 +340,7 @@ define i32 @anyof_blend_with_select_mask(i1 %a, i1 %b, i32 %n) {
; CHECK-NEXT: [[EC:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
; CHECK-NEXT: br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP9:![0-9]+]]
; CHECK: [[EXIT]]:
-; CHECK-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ 0, %[[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ [[RDX_SELECT]], %[[MIDDLE_BLOCK]] ]
; CHECK-NEXT: ret i32 [[RDX_NEXT_LCSSA]]
;
entry:
@@ -309,7 +354,7 @@ loop.header:
if.then:
%cmp.inner = icmp eq i32 %iv, 0
- %sel = select i1 %cmp.inner, i32 0, i32 %rdx
+ %sel = select i1 %cmp.inner, i32 1, i32 %rdx
br label %loop.latch
loop.latch:
More information about the llvm-commits
mailing list