[llvm] [LV] - Fix crash in epilogue vectorization when AnyOf reduction has no ComputeReductionResult. (PR #213632)
Pawan Nirpal via llvm-commits
llvm-commits at lists.llvm.org
Mon Aug 24 03:23:28 PDT 2026
https://github.com/pawan-nirpal-031 updated https://github.com/llvm/llvm-project/pull/213632
>From ce6347d4e088fdc2941e1a2e267cf04b75d6276b Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Tue, 16 Dec 2025 11:07:23 +0530
Subject: [PATCH 1/5] [AArch64] - Allow for aggressive unrolling, with non-zero
LoopMicroOpBufferSize for Oryon
---
llvm/lib/Target/AArch64/AArch64SchedOryon.td | 2 +-
.../aarch64-mcpu-oryon-runtime-unroll.ll | 152 ++++++++++++++++++
2 files changed, 153 insertions(+), 1 deletion(-)
create mode 100644 llvm/test/CodeGen/AArch64/aarch64-mcpu-oryon-runtime-unroll.ll
diff --git a/llvm/lib/Target/AArch64/AArch64SchedOryon.td b/llvm/lib/Target/AArch64/AArch64SchedOryon.td
index 5b597b91e7459..435eaf99c6175 100644
--- a/llvm/lib/Target/AArch64/AArch64SchedOryon.td
+++ b/llvm/lib/Target/AArch64/AArch64SchedOryon.td
@@ -19,7 +19,7 @@ def OryonModel : SchedMachineModel {
let MicroOpBufferSize = 376;
let LoadLatency = 4;
let MispredictPenalty = 13; // 13 cycles for mispredicted branch.
- let LoopMicroOpBufferSize = 0; // Do not have a LoopMicroOpBuffer
+ let LoopMicroOpBufferSize = 16; // Oryon-1 does not have loop micro op buffer, we enable this pseudo value to allow for aggressive unrolling based on runtime TC.
let PostRAScheduler = 1; // Using PostRA sched.
let CompleteModel = 1;
diff --git a/llvm/test/CodeGen/AArch64/aarch64-mcpu-oryon-runtime-unroll.ll b/llvm/test/CodeGen/AArch64/aarch64-mcpu-oryon-runtime-unroll.ll
new file mode 100644
index 0000000000000..79136cf71c005
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/aarch64-mcpu-oryon-runtime-unroll.ll
@@ -0,0 +1,152 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes='loop-unroll' -S %s | FileCheck %s --check-prefix=UNROLLED
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @foo(ptr %mat, ptr %sharr, ptr %barr, i16 %rows, i16 %dimout) #0 {
+; UNROLLED-LABEL: define void @foo(
+; UNROLLED-SAME: ptr [[MAT:%.*]], ptr [[SHARR:%.*]], ptr [[BARR:%.*]], i16 [[ROWS:%.*]], i16 [[DIMOUT:%.*]]) #[[ATTR0:[0-9]+]] {
+; UNROLLED-NEXT: [[ENTRY:.*:]]
+; UNROLLED-NEXT: [[CMP33:%.*]] = icmp sgt i16 [[DIMOUT]], 0
+; UNROLLED-NEXT: br i1 [[CMP33]], label %[[FOR_BODY_LR_PH:.*]], label %[[FOR_END22:.*]]
+; UNROLLED: [[FOR_BODY_LR_PH]]:
+; UNROLLED-NEXT: [[CMP631:%.*]] = icmp sgt i16 [[ROWS]], 0
+; UNROLLED-NEXT: br i1 [[CMP631]], label %[[FOR_BODY_US_PREHEADER:.*]], label %[[FOR_BODY_LR_PH_SPLIT:.*]]
+; UNROLLED: [[FOR_BODY_US_PREHEADER]]:
+; UNROLLED-NEXT: [[WIDE_TRIP_COUNT39:%.*]] = zext nneg i16 [[DIMOUT]] to i64
+; UNROLLED-NEXT: [[WIDE_TRIP_COUNT:%.*]] = zext nneg i16 [[ROWS]] to i64
+; UNROLLED-NEXT: [[TMP0:%.*]] = add nsw i64 [[WIDE_TRIP_COUNT]], -1
+; UNROLLED-NEXT: br label %[[FOR_BODY_US:.*]]
+; UNROLLED: [[FOR_BODY_US]]:
+; UNROLLED-NEXT: [[INDVARS_IV36:%.*]] = phi i64 [ 0, %[[FOR_BODY_US_PREHEADER]] ], [ [[INDVARS_IV_NEXT37:%.*]], %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US:.*]] ]
+; UNROLLED-NEXT: store i8 0, ptr [[BARR]], align 1
+; UNROLLED-NEXT: [[INVARIANT_GEP_US:%.*]] = getelementptr i8, ptr [[MAT]], i64 [[INDVARS_IV36]]
+; UNROLLED-NEXT: [[XTRAITER:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 1
+; UNROLLED-NEXT: [[TMP1:%.*]] = icmp ult i64 [[TMP0]], 1
+; UNROLLED-NEXT: br i1 [[TMP1]], label %[[FOR_BODY8_US_EPIL_PREHEADER:.*]], label %[[FOR_BODY_US_NEW:.*]]
+; UNROLLED: [[FOR_BODY_US_NEW]]:
+; UNROLLED-NEXT: [[UNROLL_ITER:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[XTRAITER]]
+; UNROLLED-NEXT: br label %[[FOR_BODY8_US:.*]]
+; UNROLLED: [[FOR_BODY8_US]]:
+; UNROLLED-NEXT: [[INDVARS_IV:%.*]] = phi i64 [ 0, %[[FOR_BODY_US_NEW]] ], [ [[INDVARS_IV_NEXT_1:%.*]], %[[FOR_INC_US_1:.*]] ]
+; UNROLLED-NEXT: [[TMP2:%.*]] = phi i8 [ 0, %[[FOR_BODY_US_NEW]] ], [ [[TMP8:%.*]], %[[FOR_INC_US_1]] ]
+; UNROLLED-NEXT: [[NITER:%.*]] = phi i64 [ 0, %[[FOR_BODY_US_NEW]] ], [ [[NITER_NEXT_1:%.*]], %[[FOR_INC_US_1]] ]
+; UNROLLED-NEXT: [[GEP_US:%.*]] = getelementptr [2 x i8], ptr [[INVARIANT_GEP_US]], i64 [[INDVARS_IV]]
+; UNROLLED-NEXT: [[TMP3:%.*]] = load i8, ptr [[GEP_US]], align 1
+; UNROLLED-NEXT: [[TOBOOL_NOT_US:%.*]] = icmp eq i8 [[TMP3]], 0
+; UNROLLED-NEXT: br i1 [[TOBOOL_NOT_US]], label %[[FOR_INC_US:.*]], label %[[IF_THEN_US:.*]]
+; UNROLLED: [[IF_THEN_US]]:
+; UNROLLED-NEXT: [[ARRAYIDX14_US:%.*]] = getelementptr inbounds nuw i8, ptr [[SHARR]], i64 [[INDVARS_IV]]
+; UNROLLED-NEXT: [[TMP4:%.*]] = load i8, ptr [[ARRAYIDX14_US]], align 1
+; UNROLLED-NEXT: [[XOR30_US:%.*]] = xor i8 [[TMP2]], [[TMP4]]
+; UNROLLED-NEXT: store i8 [[XOR30_US]], ptr [[BARR]], align 1
+; UNROLLED-NEXT: br label %[[FOR_INC_US]]
+; UNROLLED: [[FOR_INC_US]]:
+; UNROLLED-NEXT: [[TMP5:%.*]] = phi i8 [ [[TMP2]], %[[FOR_BODY8_US]] ], [ [[XOR30_US]], %[[IF_THEN_US]] ]
+; UNROLLED-NEXT: [[INDVARS_IV_NEXT:%.*]] = add nuw nsw i64 [[INDVARS_IV]], 1
+; UNROLLED-NEXT: [[GEP_US_1:%.*]] = getelementptr [2 x i8], ptr [[INVARIANT_GEP_US]], i64 [[INDVARS_IV_NEXT]]
+; UNROLLED-NEXT: [[TMP6:%.*]] = load i8, ptr [[GEP_US_1]], align 1
+; UNROLLED-NEXT: [[TOBOOL_NOT_US_1:%.*]] = icmp eq i8 [[TMP6]], 0
+; UNROLLED-NEXT: br i1 [[TOBOOL_NOT_US_1]], label %[[FOR_INC_US_1]], label %[[IF_THEN_US_1:.*]]
+; UNROLLED: [[IF_THEN_US_1]]:
+; UNROLLED-NEXT: [[ARRAYIDX14_US_1:%.*]] = getelementptr inbounds nuw i8, ptr [[SHARR]], i64 [[INDVARS_IV_NEXT]]
+; UNROLLED-NEXT: [[TMP7:%.*]] = load i8, ptr [[ARRAYIDX14_US_1]], align 1
+; UNROLLED-NEXT: [[XOR30_US_1:%.*]] = xor i8 [[TMP5]], [[TMP7]]
+; UNROLLED-NEXT: store i8 [[XOR30_US_1]], ptr [[BARR]], align 1
+; UNROLLED-NEXT: br label %[[FOR_INC_US_1]]
+; UNROLLED: [[FOR_INC_US_1]]:
+; UNROLLED-NEXT: [[TMP8]] = phi i8 [ [[TMP5]], %[[FOR_INC_US]] ], [ [[XOR30_US_1]], %[[IF_THEN_US_1]] ]
+; UNROLLED-NEXT: [[INDVARS_IV_NEXT_1]] = add nuw nsw i64 [[INDVARS_IV]], 2
+; UNROLLED-NEXT: [[NITER_NEXT_1]] = add i64 [[NITER]], 2
+; UNROLLED-NEXT: [[NITER_NCMP_1:%.*]] = icmp eq i64 [[NITER_NEXT_1]], [[UNROLL_ITER]]
+; UNROLLED-NEXT: br i1 [[NITER_NCMP_1]], label %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US_UNR_LCSSA:.*]], label %[[FOR_BODY8_US]]
+; UNROLLED: [[FOR_COND3_FOR_INC20_CRIT_EDGE_US_UNR_LCSSA]]:
+; UNROLLED-NEXT: [[INDVARS_IV_UNR:%.*]] = phi i64 [ [[INDVARS_IV_NEXT_1]], %[[FOR_INC_US_1]] ]
+; UNROLLED-NEXT: [[DOTUNR:%.*]] = phi i8 [ [[TMP8]], %[[FOR_INC_US_1]] ]
+; UNROLLED-NEXT: [[LCMP_MOD:%.*]] = icmp ne i64 [[XTRAITER]], 0
+; UNROLLED-NEXT: br i1 [[LCMP_MOD]], label %[[FOR_BODY8_US_EPIL_PREHEADER]], label %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US]]
+; UNROLLED: [[FOR_BODY8_US_EPIL_PREHEADER]]:
+; UNROLLED-NEXT: [[INDVARS_IV_EPIL_INIT:%.*]] = phi i64 [ 0, %[[FOR_BODY_US]] ], [ [[INDVARS_IV_UNR]], %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US_UNR_LCSSA]] ]
+; UNROLLED-NEXT: [[DOTEPIL_INIT:%.*]] = phi i8 [ 0, %[[FOR_BODY_US]] ], [ [[DOTUNR]], %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US_UNR_LCSSA]] ]
+; UNROLLED-NEXT: [[LCMP_MOD1:%.*]] = icmp ne i64 [[XTRAITER]], 0
+; UNROLLED-NEXT: call void @llvm.assume(i1 [[LCMP_MOD1]])
+; UNROLLED-NEXT: br label %[[FOR_BODY8_US_EPIL:.*]]
+; UNROLLED: [[FOR_BODY8_US_EPIL]]:
+; UNROLLED-NEXT: [[GEP_US_EPIL:%.*]] = getelementptr [2 x i8], ptr [[INVARIANT_GEP_US]], i64 [[INDVARS_IV_EPIL_INIT]]
+; UNROLLED-NEXT: [[TMP9:%.*]] = load i8, ptr [[GEP_US_EPIL]], align 1
+; UNROLLED-NEXT: [[TOBOOL_NOT_US_EPIL:%.*]] = icmp eq i8 [[TMP9]], 0
+; UNROLLED-NEXT: br i1 [[TOBOOL_NOT_US_EPIL]], label %[[FOR_INC_US_EPIL:.*]], label %[[IF_THEN_US_EPIL:.*]]
+; UNROLLED: [[IF_THEN_US_EPIL]]:
+; UNROLLED-NEXT: [[ARRAYIDX14_US_EPIL:%.*]] = getelementptr inbounds nuw i8, ptr [[SHARR]], i64 [[INDVARS_IV_EPIL_INIT]]
+; UNROLLED-NEXT: [[TMP10:%.*]] = load i8, ptr [[ARRAYIDX14_US_EPIL]], align 1
+; UNROLLED-NEXT: [[XOR30_US_EPIL:%.*]] = xor i8 [[DOTEPIL_INIT]], [[TMP10]]
+; UNROLLED-NEXT: store i8 [[XOR30_US_EPIL]], ptr [[BARR]], align 1
+; UNROLLED-NEXT: br label %[[FOR_INC_US_EPIL]]
+; UNROLLED: [[FOR_INC_US_EPIL]]:
+; UNROLLED-NEXT: br label %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US]]
+; UNROLLED: [[FOR_COND3_FOR_INC20_CRIT_EDGE_US]]:
+; UNROLLED-NEXT: [[INDVARS_IV_NEXT37]] = add nuw nsw i64 [[INDVARS_IV36]], 1
+; UNROLLED-NEXT: [[EXITCOND40_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT37]], [[WIDE_TRIP_COUNT39]]
+; UNROLLED-NEXT: br i1 [[EXITCOND40_NOT]], label %[[FOR_END22_LOOPEXIT:.*]], label %[[FOR_BODY_US]]
+; UNROLLED: [[FOR_BODY_LR_PH_SPLIT]]:
+; UNROLLED-NEXT: store i8 0, ptr [[BARR]], align 1
+; UNROLLED-NEXT: br label %[[FOR_END22]]
+; UNROLLED: [[FOR_END22_LOOPEXIT]]:
+; UNROLLED-NEXT: br label %[[FOR_END22]]
+; UNROLLED: [[FOR_END22]]:
+; UNROLLED-NEXT: ret void
+;
+entry:
+ %cmp33 = icmp sgt i16 %dimout, 0
+ br i1 %cmp33, label %for.body.lr.ph, label %for.end22
+
+for.body.lr.ph: ; preds = %entry
+ %cmp631 = icmp sgt i16 %rows, 0
+ br i1 %cmp631, label %for.body.us.preheader, label %for.body.lr.ph.split
+
+for.body.us.preheader: ; preds = %for.body.lr.ph
+ %wide.trip.count39 = zext nneg i16 %dimout to i64
+ %wide.trip.count = zext nneg i16 %rows to i64
+ br label %for.body.us
+
+for.body.us: ; preds = %for.body.us.preheader, %for.cond3.for.inc20_crit_edge.us
+ %indvars.iv36 = phi i64 [ 0, %for.body.us.preheader ], [ %indvars.iv.next37, %for.cond3.for.inc20_crit_edge.us ]
+ store i8 0, ptr %barr, align 1
+ %invariant.gep.us = getelementptr i8, ptr %mat, i64 %indvars.iv36
+ br label %for.body8.us
+
+for.body8.us: ; preds = %for.body.us, %for.inc.us
+ %indvars.iv = phi i64 [ 0, %for.body.us ], [ %indvars.iv.next, %for.inc.us ]
+ %0 = phi i8 [ 0, %for.body.us ], [ %3, %for.inc.us ]
+ %gep.us = getelementptr [2 x i8], ptr %invariant.gep.us, i64 %indvars.iv
+ %1 = load i8, ptr %gep.us, align 1
+ %tobool.not.us = icmp eq i8 %1, 0
+ br i1 %tobool.not.us, label %for.inc.us, label %if.then.us
+
+if.then.us: ; preds = %for.body8.us
+ %arrayidx14.us = getelementptr inbounds nuw i8, ptr %sharr, i64 %indvars.iv
+ %2 = load i8, ptr %arrayidx14.us, align 1
+ %xor30.us = xor i8 %0, %2
+ store i8 %xor30.us, ptr %barr, align 1
+ br label %for.inc.us
+
+for.inc.us: ; preds = %if.then.us, %for.body8.us
+ %3 = phi i8 [ %0, %for.body8.us ], [ %xor30.us, %if.then.us ]
+ %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+ %exitcond.not = icmp eq i64 %indvars.iv.next, %wide.trip.count
+ br i1 %exitcond.not, label %for.cond3.for.inc20_crit_edge.us, label %for.body8.us
+
+for.cond3.for.inc20_crit_edge.us: ; preds = %for.inc.us
+ %indvars.iv.next37 = add nuw nsw i64 %indvars.iv36, 1
+ %exitcond40.not = icmp eq i64 %indvars.iv.next37, %wide.trip.count39
+ br i1 %exitcond40.not, label %for.end22, label %for.body.us
+
+for.body.lr.ph.split: ; preds = %for.body.lr.ph
+ store i8 0, ptr %barr, align 1
+ br label %for.end22
+
+for.end22: ; preds = %for.cond3.for.inc20_crit_edge.us, %for.body.lr.ph.split, %entry
+ ret void
+}
+
+attributes #0 = { "target-cpu"="oryon-1" "target-features"="+neon,+sve" }
>From e5f3bb148fa0d2a15d9c53e331d9723ef2ff4a7e Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Mon, 3 Aug 2026 02:25:55 -0700
Subject: [PATCH 2/5] [LV] - Fix crash in epilogue vectorization when AnyOf
reduction has no ComputeReductionResult
---
.../Transforms/Vectorize/LoopVectorize.cpp | 13 +++++---
...pilog-vectorization-anyof-no-rdx-result.ll | 31 +++++++++++++++++++
2 files changed, 40 insertions(+), 4 deletions(-)
create mode 100644 llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index de385ae0a4511..64ef070d1c84e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7624,14 +7624,19 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
// TODO: Move setting of resume values to prepareToExecute.
if (auto *ReductionPhi = dyn_cast<VPReductionPHIRecipe>(&R)) {
// Find the reduction result by searching users of the phi or its backedge
- // value.
+ // value, looking through intermediate recipes.
auto IsReductionResult = [](VPRecipeBase *R) {
auto *VPI = dyn_cast<VPInstruction>(R);
return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
};
- auto *RdxResult = cast<VPInstruction>(
- vputils::findRecipe(ReductionPhi->getBackedgeValue(), IsReductionResult));
- assert(RdxResult && "expected to find reduction result");
+ auto *RdxResult = dyn_cast_or_null<VPInstruction>(vputils::findRecipe(
+ ReductionPhi->getBackedgeValue(), IsReductionResult));
+ // If the ComputeReductionResult was optimized away (e.g., the exit value
+ // was simplified to the start value), the reduction does not contribute
+ // to the exit value. Skip updating the start value for the epilogue,
+ // keeping it as the identity value.
+ if (!RdxResult)
+ continue;
VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
cast<PHINode>(ReductionPhi->getUnderlyingInstr()));
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
new file mode 100644
index 0000000000000..d7fea690e4692
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -0,0 +1,31 @@
+; RUN: opt -passes=loop-vectorize -S %s | FileCheck %s
+;
+; Verify that epilogue vectorization does not crash when a VPReductionPHIRecipe
+; (AnyOf reduction) has no ComputeReductionResult because the exit value was
+; simplified to a constant.
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+; CHECK-LABEL: @main
+; CHECK: vec.epilog.vector.body:
+; CHECK: middle.block:
+define i8 @main() {
+entry:
+ br label %loop
+
+exit:
+ ret i8 %sel
+
+loop:
+ %phi.rdx = phi i8 [ %sel, %loop ], [ 0, %entry ]
+ %iv = phi i32 [ %iv.next, %loop ], [ 1, %entry ]
+ %cmp = icmp sgt i32 0, 0
+ %sel = select i1 %cmp, i8 0, i8 %phi.rdx
+ %iv.next = add i32 %iv, 1
+ %exit.cond = icmp eq i32 %iv.next, 0
+ br i1 %exit.cond, label %exit, label %loop
+
+; uselistorder directives
+ uselistorder i8 %sel, { 1, 0 }
+}
>From 05514df72acddbc28318cc82858827d0748ebe24 Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pawannirpal at gmail.com>
Date: Mon, 3 Aug 2026 15:09:50 +0530
Subject: [PATCH 3/5] Update epilog-vectorization-anyof-no-rdx-result.ll
---
.../LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll | 3 ---
1 file changed, 3 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
index d7fea690e4692..efaeeca1169c8 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -1,8 +1,5 @@
; RUN: opt -passes=loop-vectorize -S %s | FileCheck %s
;
-; Verify that epilogue vectorization does not crash when a VPReductionPHIRecipe
-; (AnyOf reduction) has no ComputeReductionResult because the exit value was
-; simplified to a constant.
target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
target triple = "x86_64-unknown-linux-gnu"
>From ebaaceea6d4774950932e1fdb7b97cfae7aabf52 Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Mon, 3 Aug 2026 02:25:55 -0700
Subject: [PATCH 4/5] [LV] - Fix crash in epilogue vectorization when AnyOf
reduction has no ComputeReductionResult
---
.../Transforms/Vectorize/LoopVectorize.cpp | 20 ++++--
...pilog-vectorization-anyof-no-rdx-result.ll | 70 ++++++++++++++++---
2 files changed, 73 insertions(+), 17 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 64ef070d1c84e..61813bd7bc8c6 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7619,7 +7619,7 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
SmallVector<Instruction *> InstsToMove;
// Ensure that the start values for all header phi recipes are updated before
// vectorizing the epilogue loop.
- for (VPRecipeBase &R : Header->phis()) {
+ for (VPRecipeBase &R : make_early_inc_range(Header->phis())) {
Value *ResumeV = nullptr;
// TODO: Move setting of resume values to prepareToExecute.
if (auto *ReductionPhi = dyn_cast<VPReductionPHIRecipe>(&R)) {
@@ -7629,14 +7629,22 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
auto *VPI = dyn_cast<VPInstruction>(R);
return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
};
- auto *RdxResult = dyn_cast_or_null<VPInstruction>(vputils::findRecipe(
- ReductionPhi->getBackedgeValue(), IsReductionResult));
+ VPValue *BackedgeVal = ReductionPhi->getBackedgeValue();
+ auto *RdxResult = cast_or_null<VPInstruction>(
+ vputils::findRecipe(BackedgeVal, IsReductionResult));
// If the ComputeReductionResult was optimized away (e.g., the exit value
// was simplified to the start value), the reduction does not contribute
- // to the exit value. Skip updating the start value for the epilogue,
- // keeping it as the identity value.
- if (!RdxResult)
+ // to the exit value. Get rid of the dead reduction cycle.
+ if (!RdxResult) {
+ ReductionPhi->replaceAllUsesWith(ReductionPhi->getStartValue());
+ // The backedge value may be the phi itself (if the backedge chain was
+ // simplified away), or a separate recipe forming a cycle with the phi.
+ bool BackedgeIsPhi = (BackedgeVal == ReductionPhi);
+ ReductionPhi->eraseFromParent();
+ if (!BackedgeIsPhi)
+ vputils::recursivelyDeleteDeadRecipes(BackedgeVal);
continue;
+ }
VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
cast<PHINode>(ReductionPhi->getUnderlyingInstr()));
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
index efaeeca1169c8..f80dce8cbafc5 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -1,13 +1,56 @@
-; RUN: opt -passes=loop-vectorize -S %s | FileCheck %s
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN: -enable-epilogue-vectorization -epilogue-vectorization-force-VF=2 -S %s \
+; RUN: | FileCheck %s
;
+; Verify that epilogue vectorization does not crash when a VPReductionPHIRecipe
+; (AnyOf reduction) has no ComputeReductionResult because the exit value was
+; simplified to a constant. The dead reduction cycle should be deleted.
-target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
-target triple = "x86_64-unknown-linux-gnu"
-
-; CHECK-LABEL: @main
-; CHECK: vec.epilog.vector.body:
-; CHECK: middle.block:
-define i8 @main() {
+define i8 @dead_anyof_reduction() {
+; CHECK-LABEL: define i8 @dead_anyof_reduction() {
+; CHECK-NEXT: [[ITER_CHECK:.*]]:
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[VEC_PHI]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], -4
+; CHECK-NEXT: br i1 [[TMP0]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK: [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT: br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
+; CHECK: [[VEC_EPILOG_PH]]:
+; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ -4, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK: [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX1:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT2:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT2]] = add nuw i32 [[INDEX1]], 2
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq i32 [[INDEX_NEXT2]], -2
+; CHECK-NEXT: br i1 [[TMP1]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT: br i1 false, label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT: [[BC_MERGE_RDX3:%.*]] = phi i8 [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT: [[BC_RESUME_VAL4:%.*]] = phi i32 [ -1, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ -3, %[[VEC_EPILOG_ITER_CHECK]] ], [ 1, %[[ITER_CHECK]] ]
+; CHECK-NEXT: br label %[[LOOP:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[SEL_LCSSA:%.*]] = phi i8 [ [[SEL:%.*]], %[[LOOP]] ], [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i8 [[SEL_LCSSA]]
+; CHECK: [[LOOP]]:
+; CHECK-NEXT: [[PHI_RDX:%.*]] = phi i8 [ [[SEL]], %[[LOOP]] ], [ [[BC_MERGE_RDX3]], %[[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[LOOP]] ], [ [[BC_RESUME_VAL4]], %[[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-NEXT: [[CMP:%.*]] = icmp sgt i32 0, 0
+; CHECK-NEXT: [[SEL]] = select i1 [[CMP]], i8 0, i8 [[PHI_RDX]]
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT: [[EXIT_COND:%.*]] = icmp eq i32 [[IV_NEXT]], 0
+; CHECK-NEXT: br i1 [[EXIT_COND]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+;
entry:
br label %loop
@@ -22,7 +65,12 @@ loop:
%iv.next = add i32 %iv, 1
%exit.cond = icmp eq i32 %iv.next, 0
br i1 %exit.cond, label %exit, label %loop
-
-; uselistorder directives
- uselistorder i8 %sel, { 1, 0 }
}
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[PROF3]] = !{!"branch_weights", i32 2, i32 2}
+; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META1]], [[META2]]}
+; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META2]], [[META1]]}
+;.
>From f20d08fd68a1e85d0a7c51da2a0b365368880f31 Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Mon, 3 Aug 2026 02:25:55 -0700
Subject: [PATCH 5/5] [LV] - Fix crash in epilogue vectorization when AnyOf
reduction has no ComputeReductionResult
---
.../Transforms/Vectorize/LoopVectorize.cpp | 23 +----
.../Transforms/Vectorize/VPlanTransforms.cpp | 43 ++++++---
...pilog-vectorization-anyof-no-rdx-result.ll | 93 +++++++++++++++++--
.../LoopVectorize/select-cmp-blend-chain.ll | 33 -------
4 files changed, 117 insertions(+), 75 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 61813bd7bc8c6..de385ae0a4511 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7619,32 +7619,19 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
SmallVector<Instruction *> InstsToMove;
// Ensure that the start values for all header phi recipes are updated before
// vectorizing the epilogue loop.
- for (VPRecipeBase &R : make_early_inc_range(Header->phis())) {
+ for (VPRecipeBase &R : Header->phis()) {
Value *ResumeV = nullptr;
// TODO: Move setting of resume values to prepareToExecute.
if (auto *ReductionPhi = dyn_cast<VPReductionPHIRecipe>(&R)) {
// Find the reduction result by searching users of the phi or its backedge
- // value, looking through intermediate recipes.
+ // value.
auto IsReductionResult = [](VPRecipeBase *R) {
auto *VPI = dyn_cast<VPInstruction>(R);
return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
};
- VPValue *BackedgeVal = ReductionPhi->getBackedgeValue();
- auto *RdxResult = cast_or_null<VPInstruction>(
- vputils::findRecipe(BackedgeVal, IsReductionResult));
- // If the ComputeReductionResult was optimized away (e.g., the exit value
- // was simplified to the start value), the reduction does not contribute
- // to the exit value. Get rid of the dead reduction cycle.
- if (!RdxResult) {
- ReductionPhi->replaceAllUsesWith(ReductionPhi->getStartValue());
- // The backedge value may be the phi itself (if the backedge chain was
- // simplified away), or a separate recipe forming a cycle with the phi.
- bool BackedgeIsPhi = (BackedgeVal == ReductionPhi);
- ReductionPhi->eraseFromParent();
- if (!BackedgeIsPhi)
- vputils::recursivelyDeleteDeadRecipes(BackedgeVal);
- continue;
- }
+ auto *RdxResult = cast<VPInstruction>(
+ vputils::findRecipe(ReductionPhi->getBackedgeValue(), IsReductionResult));
+ assert(RdxResult && "expected to find reduction result");
VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
cast<PHINode>(ReductionPhi->getUnderlyingInstr()));
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 83e23449df5f5..9451d7ba71de1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -680,6 +680,32 @@ static void removeRedundantInductionCasts(VPlan &Plan) {
}
}
+/// If R is a phi-like recipe starting a dead cycle of recipes, erase all
+/// reachable recipes of the dead cycle.
+static void tryToRemoveDeadCycle(VPRecipeBase *R) {
+ auto *PhiR = dyn_cast<VPSingleDefRecipe>(R);
+ if (!PhiR || !isa<VPPhiAccessors>(R) || isa<VPCurrentIterationPHIRecipe>(R))
+ return;
+
+ // The transitive users of PhiR are closed under users, so the cycle is dead
+ // if every one of them can be erased.
+ for (VPUser *U : vputils::collectUsersRecursively(PhiR)) {
+ auto *R = cast<VPRecipeBase>(U);
+ // Bail out if a user must be retained, or if it is a phi-like recipe other
+ // than PhiR;
+ if (R->mayHaveSideEffects() || (R != PhiR && isa<VPPhiAccessors>(R)))
+ return;
+ }
+
+ // Break the cycle by replacing PhiR with its first incoming value, which is
+ // defined outside the cycle. That leaves the rest of the cycle dead.
+ PhiR->replaceAllUsesWith(PhiR->getOperand(0));
+ SmallVector<VPValue *> Incoming(PhiR->operands());
+ PhiR->eraseFromParent();
+ for (VPValue *Op : Incoming)
+ vputils::recursivelyDeleteDeadRecipes(Op);
+}
+
void VPlanTransforms::removeDeadRecipes(VPlan &Plan) {
PostOrderTraversal<VPBlockDeepTraversalWrapper<VPBlockBase *>> POT(
Plan.getEntry());
@@ -692,20 +718,9 @@ void VPlanTransforms::removeDeadRecipes(VPlan &Plan) {
continue;
}
- // Check if R is a dead VPPhi <-> update cycle and remove it.
- VPValue *Start, *Incoming;
- if (!match(&R, m_VPPhi(m_VPValue(Start), m_VPValue(Incoming))))
- continue;
- auto *PhiR = cast<VPPhi>(&R);
- VPUser *PhiUser = PhiR->getSingleUser();
- if (!PhiUser)
- continue;
- if (PhiUser != Incoming->getDefiningRecipe() ||
- Incoming->getNumUsers() != 1)
- continue;
- PhiR->replaceAllUsesWith(Start);
- PhiR->eraseFromParent();
- Incoming->getDefiningRecipe()->eraseFromParent();
+ // If R is a phi-like recipe starting a dead cycle of recipes, erase the
+ // whole cycle.
+ tryToRemoveDeadCycle(&R);
}
}
}
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
index f80dce8cbafc5..8c74af2508c9f 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
; RUN: -enable-epilogue-vectorization -epilogue-vectorization-force-VF=2 -S %s \
; RUN: | FileCheck %s
@@ -17,7 +17,6 @@ define i8 @dead_anyof_reduction() {
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[VEC_PHI]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
; CHECK-NEXT: [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], -4
; CHECK-NEXT: br i1 [[TMP0]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
@@ -66,11 +65,85 @@ loop:
%exit.cond = icmp eq i32 %iv.next, 0
br i1 %exit.cond, label %exit, label %loop
}
-;.
-; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
-; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
-; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
-; CHECK: [[PROF3]] = !{!"branch_weights", i32 2, i32 2}
-; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META1]], [[META2]]}
-; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META2]], [[META1]]}
-;.
+
+define i32 @dead_anyof_reduction_blend(i1 %c0, i32 %n) {
+; CHECK-LABEL: define i32 @dead_anyof_reduction_blend(
+; CHECK-SAME: i1 [[C0:%.*]], i32 [[N:%.*]]) {
+; CHECK-NEXT: [[ITER_CHECK:.*]]:
+; CHECK-NEXT: [[SMAX:%.*]] = call i32 @llvm.smax.i32(i32 [[N]], i32 1)
+; CHECK-NEXT: [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[SMAX]], 2
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK: [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT: [[MIN_ITERS_CHECK1:%.*]] = icmp ult i32 [[SMAX]], 4
+; CHECK-NEXT: br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[TMP0:%.*]] = urem i32 [[SMAX]], 4
+; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[SMAX]], [[TMP0]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP1:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[TMP1]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N:%.*]] = icmp eq i32 [[SMAX]], [[N_VEC]]
+; CHECK-NEXT: br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK: [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT: [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP0]], 2
+; CHECK-NEXT: br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; CHECK: [[VEC_EPILOG_PH]]:
+; CHECK-NEXT: [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT: [[TMP2:%.*]] = urem i32 [[SMAX]], 2
+; CHECK-NEXT: [[N_VEC2:%.*]] = sub i32 [[SMAX]], [[TMP2]]
+; CHECK-NEXT: br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK: [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX3:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT: [[INDEX_NEXT4]] = add nuw i32 [[INDEX3]], 2
+; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT4]], [[N_VEC2]]
+; CHECK-NEXT: br i1 [[TMP3]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK: [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT: [[CMP_N5:%.*]] = icmp eq i32 [[SMAX]], [[N_VEC2]]
+; CHECK-NEXT: br i1 [[CMP_N5]], label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; CHECK: [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT: [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC2]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT: [[BC_MERGE_RDX7:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT: br label %[[LOOP_HEADER:.*]]
+; CHECK: [[LOOP_HEADER]]:
+; CHECK-NEXT: [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT: [[RDX:%.*]] = phi i32 [ [[BC_MERGE_RDX7]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[RDX_NEXT:%.*]], %[[LOOP_LATCH]] ]
+; CHECK-NEXT: br i1 [[C0]], label %[[LOOP_LATCH]], label %[[IF_THEN:.*]]
+; CHECK: [[IF_THEN]]:
+; CHECK-NEXT: [[C:%.*]] = icmp eq i32 [[IV]], 0
+; CHECK-NEXT: [[SEL:%.*]] = select i1 [[C]], i32 0, i32 [[RDX]]
+; CHECK-NEXT: br label %[[LOOP_LATCH]]
+; CHECK: [[LOOP_LATCH]]:
+; CHECK-NEXT: [[RDX_NEXT]] = phi i32 [ [[SEL]], %[[IF_THEN]] ], [ [[RDX]], %[[LOOP_HEADER]] ]
+; CHECK-NEXT: [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT: [[EC:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT: br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i32 [[RDX_NEXT_LCSSA]]
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %rdx = phi i32 [ 0, %entry ], [ %rdx.next, %loop.latch ]
+ br i1 %c0, label %loop.latch, label %if.then
+
+if.then:
+ %c = icmp eq i32 %iv, 0
+ %sel = select i1 %c, i32 0, i32 %rdx
+ br label %loop.latch
+
+loop.latch:
+ %rdx.next = phi i32 [ %sel, %if.then ], [ %rdx, %loop.header ]
+ %iv.next = add i32 %iv, 1
+ %ec = icmp slt i32 %iv.next, %n
+ br i1 %ec, label %loop.header, label %exit
+
+exit:
+ ret i32 %rdx.next
+}
diff --git a/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll b/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
index 8b12e09631246..4318798ffadda 100644
--- a/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
+++ b/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
@@ -15,14 +15,7 @@ define i32 @anyof_two_blend_chain(i1 %c0, i1 %c1, i32 %n) {
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI1:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT: [[TMP1:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP0]]
-; CHECK-NEXT: [[PREDPHI:%.*]] = select i1 [[C1]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP1]]
-; CHECK-NEXT: [[PREDPHI1]] = select i1 [[C0]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -97,15 +90,7 @@ define i32 @anyof_three_blend_chain(i1 %c0, i1 %c1, i1 %c2, i32 %n) {
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI2:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT: [[TMP1:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP0]]
-; CHECK-NEXT: [[PREDPHI:%.*]] = select i1 [[C2]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP1]]
-; CHECK-NEXT: [[PREDPHI1:%.*]] = select i1 [[C1]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI]]
-; CHECK-NEXT: [[PREDPHI2]] = select i1 [[C0]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI1]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP2:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -189,21 +174,10 @@ define i32 @anyof_diamond_blend_chain(i1 %c0, i1 %c1, i32 %n) {
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i32 [[SMAX]], 4
; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[SMAX]], [[N_MOD_VF]]
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i1> poison, i1 [[C1]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT1]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <4 x i1> poison, i1 [[C0]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT3:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT2]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP1:%.*]] = select <4 x i1> [[BROADCAST_SPLAT2]], <4 x i1> [[BROADCAST_SPLAT3]], <4 x i1> zeroinitializer
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP0:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP0]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[TMP3]] = or <4 x i1> [[VEC_PHI]], [[TMP2]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
@@ -293,17 +267,10 @@ define i32 @anyof_blend_with_select_mask(i1 %a, i1 %b, i32 %n) {
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: [[N_MOD_VF:%.*]] = urem i32 [[SMAX]], 4
; CHECK-NEXT: [[N_VEC:%.*]] = sub i32 [[SMAX]], [[N_MOD_VF]]
-; CHECK-NEXT: [[TMP0:%.*]] = select i1 [[A]], i1 [[B]], i1 false
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP1:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP1]]
-; CHECK-NEXT: [[PREDPHI]] = select i1 [[TMP0]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP2]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT: [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
; CHECK-NEXT: br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
More information about the llvm-commits
mailing list