[llvm] [VPlan] - Generalize dead cycle removal in removeDeadRecipes. (PR #213632)

Pawan Nirpal via llvm-commits llvm-commits at lists.llvm.org
Mon Aug 24 07:05:30 PDT 2026


https://github.com/pawan-nirpal-031 updated https://github.com/llvm/llvm-project/pull/213632

>From ce6347d4e088fdc2941e1a2e267cf04b75d6276b Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Tue, 16 Dec 2025 11:07:23 +0530
Subject: [PATCH 1/5] [AArch64] - Allow for aggressive unrolling, with non-zero
 LoopMicroOpBufferSize for Oryon

---
 llvm/lib/Target/AArch64/AArch64SchedOryon.td  |   2 +-
 .../aarch64-mcpu-oryon-runtime-unroll.ll      | 152 ++++++++++++++++++
 2 files changed, 153 insertions(+), 1 deletion(-)
 create mode 100644 llvm/test/CodeGen/AArch64/aarch64-mcpu-oryon-runtime-unroll.ll

diff --git a/llvm/lib/Target/AArch64/AArch64SchedOryon.td b/llvm/lib/Target/AArch64/AArch64SchedOryon.td
index 5b597b91e7459..435eaf99c6175 100644
--- a/llvm/lib/Target/AArch64/AArch64SchedOryon.td
+++ b/llvm/lib/Target/AArch64/AArch64SchedOryon.td
@@ -19,7 +19,7 @@ def OryonModel : SchedMachineModel {
   let MicroOpBufferSize     = 376;
   let LoadLatency           =   4;
   let MispredictPenalty     =  13; // 13 cycles for mispredicted branch.
-  let LoopMicroOpBufferSize =   0; // Do not have a LoopMicroOpBuffer
+  let LoopMicroOpBufferSize =   16; // Oryon-1 does not have loop micro op buffer, we enable this pseudo value to allow for aggressive unrolling based on runtime TC.
   let PostRAScheduler       =   1; // Using PostRA sched.
   let CompleteModel         =   1;
 
diff --git a/llvm/test/CodeGen/AArch64/aarch64-mcpu-oryon-runtime-unroll.ll b/llvm/test/CodeGen/AArch64/aarch64-mcpu-oryon-runtime-unroll.ll
new file mode 100644
index 0000000000000..79136cf71c005
--- /dev/null
+++ b/llvm/test/CodeGen/AArch64/aarch64-mcpu-oryon-runtime-unroll.ll
@@ -0,0 +1,152 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes='loop-unroll' -S %s | FileCheck %s --check-prefix=UNROLLED
+
+target triple = "aarch64-unknown-linux-gnu"
+
+define void @foo(ptr %mat, ptr %sharr, ptr %barr, i16 %rows, i16 %dimout) #0 {
+; UNROLLED-LABEL: define void @foo(
+; UNROLLED-SAME: ptr [[MAT:%.*]], ptr [[SHARR:%.*]], ptr [[BARR:%.*]], i16 [[ROWS:%.*]], i16 [[DIMOUT:%.*]]) #[[ATTR0:[0-9]+]] {
+; UNROLLED-NEXT:  [[ENTRY:.*:]]
+; UNROLLED-NEXT:    [[CMP33:%.*]] = icmp sgt i16 [[DIMOUT]], 0
+; UNROLLED-NEXT:    br i1 [[CMP33]], label %[[FOR_BODY_LR_PH:.*]], label %[[FOR_END22:.*]]
+; UNROLLED:       [[FOR_BODY_LR_PH]]:
+; UNROLLED-NEXT:    [[CMP631:%.*]] = icmp sgt i16 [[ROWS]], 0
+; UNROLLED-NEXT:    br i1 [[CMP631]], label %[[FOR_BODY_US_PREHEADER:.*]], label %[[FOR_BODY_LR_PH_SPLIT:.*]]
+; UNROLLED:       [[FOR_BODY_US_PREHEADER]]:
+; UNROLLED-NEXT:    [[WIDE_TRIP_COUNT39:%.*]] = zext nneg i16 [[DIMOUT]] to i64
+; UNROLLED-NEXT:    [[WIDE_TRIP_COUNT:%.*]] = zext nneg i16 [[ROWS]] to i64
+; UNROLLED-NEXT:    [[TMP0:%.*]] = add nsw i64 [[WIDE_TRIP_COUNT]], -1
+; UNROLLED-NEXT:    br label %[[FOR_BODY_US:.*]]
+; UNROLLED:       [[FOR_BODY_US]]:
+; UNROLLED-NEXT:    [[INDVARS_IV36:%.*]] = phi i64 [ 0, %[[FOR_BODY_US_PREHEADER]] ], [ [[INDVARS_IV_NEXT37:%.*]], %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US:.*]] ]
+; UNROLLED-NEXT:    store i8 0, ptr [[BARR]], align 1
+; UNROLLED-NEXT:    [[INVARIANT_GEP_US:%.*]] = getelementptr i8, ptr [[MAT]], i64 [[INDVARS_IV36]]
+; UNROLLED-NEXT:    [[XTRAITER:%.*]] = and i64 [[WIDE_TRIP_COUNT]], 1
+; UNROLLED-NEXT:    [[TMP1:%.*]] = icmp ult i64 [[TMP0]], 1
+; UNROLLED-NEXT:    br i1 [[TMP1]], label %[[FOR_BODY8_US_EPIL_PREHEADER:.*]], label %[[FOR_BODY_US_NEW:.*]]
+; UNROLLED:       [[FOR_BODY_US_NEW]]:
+; UNROLLED-NEXT:    [[UNROLL_ITER:%.*]] = sub i64 [[WIDE_TRIP_COUNT]], [[XTRAITER]]
+; UNROLLED-NEXT:    br label %[[FOR_BODY8_US:.*]]
+; UNROLLED:       [[FOR_BODY8_US]]:
+; UNROLLED-NEXT:    [[INDVARS_IV:%.*]] = phi i64 [ 0, %[[FOR_BODY_US_NEW]] ], [ [[INDVARS_IV_NEXT_1:%.*]], %[[FOR_INC_US_1:.*]] ]
+; UNROLLED-NEXT:    [[TMP2:%.*]] = phi i8 [ 0, %[[FOR_BODY_US_NEW]] ], [ [[TMP8:%.*]], %[[FOR_INC_US_1]] ]
+; UNROLLED-NEXT:    [[NITER:%.*]] = phi i64 [ 0, %[[FOR_BODY_US_NEW]] ], [ [[NITER_NEXT_1:%.*]], %[[FOR_INC_US_1]] ]
+; UNROLLED-NEXT:    [[GEP_US:%.*]] = getelementptr [2 x i8], ptr [[INVARIANT_GEP_US]], i64 [[INDVARS_IV]]
+; UNROLLED-NEXT:    [[TMP3:%.*]] = load i8, ptr [[GEP_US]], align 1
+; UNROLLED-NEXT:    [[TOBOOL_NOT_US:%.*]] = icmp eq i8 [[TMP3]], 0
+; UNROLLED-NEXT:    br i1 [[TOBOOL_NOT_US]], label %[[FOR_INC_US:.*]], label %[[IF_THEN_US:.*]]
+; UNROLLED:       [[IF_THEN_US]]:
+; UNROLLED-NEXT:    [[ARRAYIDX14_US:%.*]] = getelementptr inbounds nuw i8, ptr [[SHARR]], i64 [[INDVARS_IV]]
+; UNROLLED-NEXT:    [[TMP4:%.*]] = load i8, ptr [[ARRAYIDX14_US]], align 1
+; UNROLLED-NEXT:    [[XOR30_US:%.*]] = xor i8 [[TMP2]], [[TMP4]]
+; UNROLLED-NEXT:    store i8 [[XOR30_US]], ptr [[BARR]], align 1
+; UNROLLED-NEXT:    br label %[[FOR_INC_US]]
+; UNROLLED:       [[FOR_INC_US]]:
+; UNROLLED-NEXT:    [[TMP5:%.*]] = phi i8 [ [[TMP2]], %[[FOR_BODY8_US]] ], [ [[XOR30_US]], %[[IF_THEN_US]] ]
+; UNROLLED-NEXT:    [[INDVARS_IV_NEXT:%.*]] = add nuw nsw i64 [[INDVARS_IV]], 1
+; UNROLLED-NEXT:    [[GEP_US_1:%.*]] = getelementptr [2 x i8], ptr [[INVARIANT_GEP_US]], i64 [[INDVARS_IV_NEXT]]
+; UNROLLED-NEXT:    [[TMP6:%.*]] = load i8, ptr [[GEP_US_1]], align 1
+; UNROLLED-NEXT:    [[TOBOOL_NOT_US_1:%.*]] = icmp eq i8 [[TMP6]], 0
+; UNROLLED-NEXT:    br i1 [[TOBOOL_NOT_US_1]], label %[[FOR_INC_US_1]], label %[[IF_THEN_US_1:.*]]
+; UNROLLED:       [[IF_THEN_US_1]]:
+; UNROLLED-NEXT:    [[ARRAYIDX14_US_1:%.*]] = getelementptr inbounds nuw i8, ptr [[SHARR]], i64 [[INDVARS_IV_NEXT]]
+; UNROLLED-NEXT:    [[TMP7:%.*]] = load i8, ptr [[ARRAYIDX14_US_1]], align 1
+; UNROLLED-NEXT:    [[XOR30_US_1:%.*]] = xor i8 [[TMP5]], [[TMP7]]
+; UNROLLED-NEXT:    store i8 [[XOR30_US_1]], ptr [[BARR]], align 1
+; UNROLLED-NEXT:    br label %[[FOR_INC_US_1]]
+; UNROLLED:       [[FOR_INC_US_1]]:
+; UNROLLED-NEXT:    [[TMP8]] = phi i8 [ [[TMP5]], %[[FOR_INC_US]] ], [ [[XOR30_US_1]], %[[IF_THEN_US_1]] ]
+; UNROLLED-NEXT:    [[INDVARS_IV_NEXT_1]] = add nuw nsw i64 [[INDVARS_IV]], 2
+; UNROLLED-NEXT:    [[NITER_NEXT_1]] = add i64 [[NITER]], 2
+; UNROLLED-NEXT:    [[NITER_NCMP_1:%.*]] = icmp eq i64 [[NITER_NEXT_1]], [[UNROLL_ITER]]
+; UNROLLED-NEXT:    br i1 [[NITER_NCMP_1]], label %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US_UNR_LCSSA:.*]], label %[[FOR_BODY8_US]]
+; UNROLLED:       [[FOR_COND3_FOR_INC20_CRIT_EDGE_US_UNR_LCSSA]]:
+; UNROLLED-NEXT:    [[INDVARS_IV_UNR:%.*]] = phi i64 [ [[INDVARS_IV_NEXT_1]], %[[FOR_INC_US_1]] ]
+; UNROLLED-NEXT:    [[DOTUNR:%.*]] = phi i8 [ [[TMP8]], %[[FOR_INC_US_1]] ]
+; UNROLLED-NEXT:    [[LCMP_MOD:%.*]] = icmp ne i64 [[XTRAITER]], 0
+; UNROLLED-NEXT:    br i1 [[LCMP_MOD]], label %[[FOR_BODY8_US_EPIL_PREHEADER]], label %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US]]
+; UNROLLED:       [[FOR_BODY8_US_EPIL_PREHEADER]]:
+; UNROLLED-NEXT:    [[INDVARS_IV_EPIL_INIT:%.*]] = phi i64 [ 0, %[[FOR_BODY_US]] ], [ [[INDVARS_IV_UNR]], %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US_UNR_LCSSA]] ]
+; UNROLLED-NEXT:    [[DOTEPIL_INIT:%.*]] = phi i8 [ 0, %[[FOR_BODY_US]] ], [ [[DOTUNR]], %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US_UNR_LCSSA]] ]
+; UNROLLED-NEXT:    [[LCMP_MOD1:%.*]] = icmp ne i64 [[XTRAITER]], 0
+; UNROLLED-NEXT:    call void @llvm.assume(i1 [[LCMP_MOD1]])
+; UNROLLED-NEXT:    br label %[[FOR_BODY8_US_EPIL:.*]]
+; UNROLLED:       [[FOR_BODY8_US_EPIL]]:
+; UNROLLED-NEXT:    [[GEP_US_EPIL:%.*]] = getelementptr [2 x i8], ptr [[INVARIANT_GEP_US]], i64 [[INDVARS_IV_EPIL_INIT]]
+; UNROLLED-NEXT:    [[TMP9:%.*]] = load i8, ptr [[GEP_US_EPIL]], align 1
+; UNROLLED-NEXT:    [[TOBOOL_NOT_US_EPIL:%.*]] = icmp eq i8 [[TMP9]], 0
+; UNROLLED-NEXT:    br i1 [[TOBOOL_NOT_US_EPIL]], label %[[FOR_INC_US_EPIL:.*]], label %[[IF_THEN_US_EPIL:.*]]
+; UNROLLED:       [[IF_THEN_US_EPIL]]:
+; UNROLLED-NEXT:    [[ARRAYIDX14_US_EPIL:%.*]] = getelementptr inbounds nuw i8, ptr [[SHARR]], i64 [[INDVARS_IV_EPIL_INIT]]
+; UNROLLED-NEXT:    [[TMP10:%.*]] = load i8, ptr [[ARRAYIDX14_US_EPIL]], align 1
+; UNROLLED-NEXT:    [[XOR30_US_EPIL:%.*]] = xor i8 [[DOTEPIL_INIT]], [[TMP10]]
+; UNROLLED-NEXT:    store i8 [[XOR30_US_EPIL]], ptr [[BARR]], align 1
+; UNROLLED-NEXT:    br label %[[FOR_INC_US_EPIL]]
+; UNROLLED:       [[FOR_INC_US_EPIL]]:
+; UNROLLED-NEXT:    br label %[[FOR_COND3_FOR_INC20_CRIT_EDGE_US]]
+; UNROLLED:       [[FOR_COND3_FOR_INC20_CRIT_EDGE_US]]:
+; UNROLLED-NEXT:    [[INDVARS_IV_NEXT37]] = add nuw nsw i64 [[INDVARS_IV36]], 1
+; UNROLLED-NEXT:    [[EXITCOND40_NOT:%.*]] = icmp eq i64 [[INDVARS_IV_NEXT37]], [[WIDE_TRIP_COUNT39]]
+; UNROLLED-NEXT:    br i1 [[EXITCOND40_NOT]], label %[[FOR_END22_LOOPEXIT:.*]], label %[[FOR_BODY_US]]
+; UNROLLED:       [[FOR_BODY_LR_PH_SPLIT]]:
+; UNROLLED-NEXT:    store i8 0, ptr [[BARR]], align 1
+; UNROLLED-NEXT:    br label %[[FOR_END22]]
+; UNROLLED:       [[FOR_END22_LOOPEXIT]]:
+; UNROLLED-NEXT:    br label %[[FOR_END22]]
+; UNROLLED:       [[FOR_END22]]:
+; UNROLLED-NEXT:    ret void
+;
+entry:
+  %cmp33 = icmp sgt i16 %dimout, 0
+  br i1 %cmp33, label %for.body.lr.ph, label %for.end22
+
+for.body.lr.ph:                                   ; preds = %entry
+  %cmp631 = icmp sgt i16 %rows, 0
+  br i1 %cmp631, label %for.body.us.preheader, label %for.body.lr.ph.split
+
+for.body.us.preheader:                            ; preds = %for.body.lr.ph
+  %wide.trip.count39 = zext nneg i16 %dimout to i64
+  %wide.trip.count = zext nneg i16 %rows to i64
+  br label %for.body.us
+
+for.body.us:                                      ; preds = %for.body.us.preheader, %for.cond3.for.inc20_crit_edge.us
+  %indvars.iv36 = phi i64 [ 0, %for.body.us.preheader ], [ %indvars.iv.next37, %for.cond3.for.inc20_crit_edge.us ]
+  store i8 0, ptr %barr, align 1
+  %invariant.gep.us = getelementptr i8, ptr %mat, i64 %indvars.iv36
+  br label %for.body8.us
+
+for.body8.us:                                     ; preds = %for.body.us, %for.inc.us
+  %indvars.iv = phi i64 [ 0, %for.body.us ], [ %indvars.iv.next, %for.inc.us ]
+  %0 = phi i8 [ 0, %for.body.us ], [ %3, %for.inc.us ]
+  %gep.us = getelementptr [2 x i8], ptr %invariant.gep.us, i64 %indvars.iv
+  %1 = load i8, ptr %gep.us, align 1
+  %tobool.not.us = icmp eq i8 %1, 0
+  br i1 %tobool.not.us, label %for.inc.us, label %if.then.us
+
+if.then.us:                                       ; preds = %for.body8.us
+  %arrayidx14.us = getelementptr inbounds nuw i8, ptr %sharr, i64 %indvars.iv
+  %2 = load i8, ptr %arrayidx14.us, align 1
+  %xor30.us = xor i8 %0, %2
+  store i8 %xor30.us, ptr %barr, align 1
+  br label %for.inc.us
+
+for.inc.us:                                       ; preds = %if.then.us, %for.body8.us
+  %3 = phi i8 [ %0, %for.body8.us ], [ %xor30.us, %if.then.us ]
+  %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+  %exitcond.not = icmp eq i64 %indvars.iv.next, %wide.trip.count
+  br i1 %exitcond.not, label %for.cond3.for.inc20_crit_edge.us, label %for.body8.us
+
+for.cond3.for.inc20_crit_edge.us:                 ; preds = %for.inc.us
+  %indvars.iv.next37 = add nuw nsw i64 %indvars.iv36, 1
+  %exitcond40.not = icmp eq i64 %indvars.iv.next37, %wide.trip.count39
+  br i1 %exitcond40.not, label %for.end22, label %for.body.us
+
+for.body.lr.ph.split:                             ; preds = %for.body.lr.ph
+  store i8 0, ptr %barr, align 1
+  br label %for.end22
+
+for.end22:                                        ; preds = %for.cond3.for.inc20_crit_edge.us, %for.body.lr.ph.split, %entry
+  ret void
+}
+
+attributes #0 = { "target-cpu"="oryon-1" "target-features"="+neon,+sve" }

>From e5f3bb148fa0d2a15d9c53e331d9723ef2ff4a7e Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Mon, 3 Aug 2026 02:25:55 -0700
Subject: [PATCH 2/5] [LV] - Fix crash in epilogue vectorization when AnyOf
 reduction has no ComputeReductionResult

---
 .../Transforms/Vectorize/LoopVectorize.cpp    | 13 +++++---
 ...pilog-vectorization-anyof-no-rdx-result.ll | 31 +++++++++++++++++++
 2 files changed, 40 insertions(+), 4 deletions(-)
 create mode 100644 llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index de385ae0a4511..64ef070d1c84e 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7624,14 +7624,19 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
     // TODO: Move setting of resume values to prepareToExecute.
     if (auto *ReductionPhi = dyn_cast<VPReductionPHIRecipe>(&R)) {
       // Find the reduction result by searching users of the phi or its backedge
-      // value.
+      // value, looking through intermediate recipes.
       auto IsReductionResult = [](VPRecipeBase *R) {
         auto *VPI = dyn_cast<VPInstruction>(R);
         return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
       };
-      auto *RdxResult = cast<VPInstruction>(
-          vputils::findRecipe(ReductionPhi->getBackedgeValue(), IsReductionResult));
-      assert(RdxResult && "expected to find reduction result");
+      auto *RdxResult = dyn_cast_or_null<VPInstruction>(vputils::findRecipe(
+          ReductionPhi->getBackedgeValue(), IsReductionResult));
+      // If the ComputeReductionResult was optimized away (e.g., the exit value
+      // was simplified to the start value), the reduction does not contribute
+      // to the exit value. Skip updating the start value for the epilogue,
+      // keeping it as the identity value.
+      if (!RdxResult)
+        continue;
 
       VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
           cast<PHINode>(ReductionPhi->getUnderlyingInstr()));
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
new file mode 100644
index 0000000000000..d7fea690e4692
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -0,0 +1,31 @@
+; RUN: opt -passes=loop-vectorize -S %s | FileCheck %s
+;
+; Verify that epilogue vectorization does not crash when a VPReductionPHIRecipe
+; (AnyOf reduction) has no ComputeReductionResult because the exit value was
+; simplified to a constant.
+
+target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
+target triple = "x86_64-unknown-linux-gnu"
+
+; CHECK-LABEL: @main
+; CHECK: vec.epilog.vector.body:
+; CHECK: middle.block:
+define i8 @main() {
+entry:
+  br label %loop
+
+exit:
+  ret i8 %sel
+
+loop:
+  %phi.rdx = phi i8 [ %sel, %loop ], [ 0, %entry ]
+  %iv = phi i32 [ %iv.next, %loop ], [ 1, %entry ]
+  %cmp = icmp sgt i32 0, 0
+  %sel = select i1 %cmp, i8 0, i8 %phi.rdx
+  %iv.next = add i32 %iv, 1
+  %exit.cond = icmp eq i32 %iv.next, 0
+  br i1 %exit.cond, label %exit, label %loop
+
+; uselistorder directives
+  uselistorder i8 %sel, { 1, 0 }
+}

>From 05514df72acddbc28318cc82858827d0748ebe24 Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pawannirpal at gmail.com>
Date: Mon, 3 Aug 2026 15:09:50 +0530
Subject: [PATCH 3/5] Update epilog-vectorization-anyof-no-rdx-result.ll

---
 .../LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll  | 3 ---
 1 file changed, 3 deletions(-)

diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
index d7fea690e4692..efaeeca1169c8 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -1,8 +1,5 @@
 ; RUN: opt -passes=loop-vectorize -S %s | FileCheck %s
 ;
-; Verify that epilogue vectorization does not crash when a VPReductionPHIRecipe
-; (AnyOf reduction) has no ComputeReductionResult because the exit value was
-; simplified to a constant.
 
 target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
 target triple = "x86_64-unknown-linux-gnu"

>From ebaaceea6d4774950932e1fdb7b97cfae7aabf52 Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Mon, 3 Aug 2026 02:25:55 -0700
Subject: [PATCH 4/5] [LV] - Fix crash in epilogue vectorization when AnyOf
 reduction has no ComputeReductionResult

---
 .../Transforms/Vectorize/LoopVectorize.cpp    | 20 ++++--
 ...pilog-vectorization-anyof-no-rdx-result.ll | 70 ++++++++++++++++---
 2 files changed, 73 insertions(+), 17 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 64ef070d1c84e..61813bd7bc8c6 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7619,7 +7619,7 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
   SmallVector<Instruction *> InstsToMove;
   // Ensure that the start values for all header phi recipes are updated before
   // vectorizing the epilogue loop.
-  for (VPRecipeBase &R : Header->phis()) {
+  for (VPRecipeBase &R : make_early_inc_range(Header->phis())) {
     Value *ResumeV = nullptr;
     // TODO: Move setting of resume values to prepareToExecute.
     if (auto *ReductionPhi = dyn_cast<VPReductionPHIRecipe>(&R)) {
@@ -7629,14 +7629,22 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
         auto *VPI = dyn_cast<VPInstruction>(R);
         return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
       };
-      auto *RdxResult = dyn_cast_or_null<VPInstruction>(vputils::findRecipe(
-          ReductionPhi->getBackedgeValue(), IsReductionResult));
+      VPValue *BackedgeVal = ReductionPhi->getBackedgeValue();
+      auto *RdxResult = cast_or_null<VPInstruction>(
+          vputils::findRecipe(BackedgeVal, IsReductionResult));
       // If the ComputeReductionResult was optimized away (e.g., the exit value
       // was simplified to the start value), the reduction does not contribute
-      // to the exit value. Skip updating the start value for the epilogue,
-      // keeping it as the identity value.
-      if (!RdxResult)
+      // to the exit value. Get rid of the dead reduction cycle.
+      if (!RdxResult) {
+        ReductionPhi->replaceAllUsesWith(ReductionPhi->getStartValue());
+        // The backedge value may be the phi itself (if the backedge chain was
+        // simplified away), or a separate recipe forming a cycle with the phi.
+        bool BackedgeIsPhi = (BackedgeVal == ReductionPhi);
+        ReductionPhi->eraseFromParent();
+        if (!BackedgeIsPhi)
+          vputils::recursivelyDeleteDeadRecipes(BackedgeVal);
         continue;
+      }
 
       VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
           cast<PHINode>(ReductionPhi->getUnderlyingInstr()));
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
index efaeeca1169c8..f80dce8cbafc5 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -1,13 +1,56 @@
-; RUN: opt -passes=loop-vectorize -S %s | FileCheck %s
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
+; RUN:   -enable-epilogue-vectorization -epilogue-vectorization-force-VF=2 -S %s \
+; RUN:   | FileCheck %s
 ;
+; Verify that epilogue vectorization does not crash when a VPReductionPHIRecipe
+; (AnyOf reduction) has no ComputeReductionResult because the exit value was
+; simplified to a constant. The dead reduction cycle should be deleted.
 
-target datalayout = "e-m:e-p270:32:32-p271:32:32-p272:64:64-i64:64-i128:128-f80:128-n8:16:32:64-S128"
-target triple = "x86_64-unknown-linux-gnu"
-
-; CHECK-LABEL: @main
-; CHECK: vec.epilog.vector.body:
-; CHECK: middle.block:
-define i8 @main() {
+define i8 @dead_anyof_reduction() {
+; CHECK-LABEL: define i8 @dead_anyof_reduction() {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[VEC_PHI]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], -4
+; CHECK-NEXT:    br i1 [[TMP0]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br i1 false, label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    br i1 false, label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ -4, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX1:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT2:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[INDEX_NEXT2]] = add nuw i32 [[INDEX1]], 2
+; CHECK-NEXT:    [[TMP1:%.*]] = icmp eq i32 [[INDEX_NEXT2]], -2
+; CHECK-NEXT:    br i1 [[TMP1]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    br i1 false, label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_MERGE_RDX3:%.*]] = phi i8 [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_RESUME_VAL4:%.*]] = phi i32 [ -1, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ -3, %[[VEC_EPILOG_ITER_CHECK]] ], [ 1, %[[ITER_CHECK]] ]
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[SEL_LCSSA:%.*]] = phi i8 [ [[SEL:%.*]], %[[LOOP]] ], [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i8 [[SEL_LCSSA]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[PHI_RDX:%.*]] = phi i8 [ [[SEL]], %[[LOOP]] ], [ [[BC_MERGE_RDX3]], %[[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[IV_NEXT:%.*]], %[[LOOP]] ], [ [[BC_RESUME_VAL4]], %[[VEC_EPILOG_SCALAR_PH]] ]
+; CHECK-NEXT:    [[CMP:%.*]] = icmp sgt i32 0, 0
+; CHECK-NEXT:    [[SEL]] = select i1 [[CMP]], i8 0, i8 [[PHI_RDX]]
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT:    [[EXIT_COND:%.*]] = icmp eq i32 [[IV_NEXT]], 0
+; CHECK-NEXT:    br i1 [[EXIT_COND]], label %[[EXIT]], label %[[LOOP]], !llvm.loop [[LOOP5:![0-9]+]]
+;
 entry:
   br label %loop
 
@@ -22,7 +65,12 @@ loop:
   %iv.next = add i32 %iv, 1
   %exit.cond = icmp eq i32 %iv.next, 0
   br i1 %exit.cond, label %exit, label %loop
-
-; uselistorder directives
-  uselistorder i8 %sel, { 1, 0 }
 }
+;.
+; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
+; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
+; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
+; CHECK: [[PROF3]] = !{!"branch_weights", i32 2, i32 2}
+; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META1]], [[META2]]}
+; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META2]], [[META1]]}
+;.

>From f20d08fd68a1e85d0a7c51da2a0b365368880f31 Mon Sep 17 00:00:00 2001
From: Pawan Nirpal <pnirpal at qti.qualcomm.com>
Date: Mon, 3 Aug 2026 02:25:55 -0700
Subject: [PATCH 5/5] [LV] - Fix crash in epilogue vectorization when AnyOf
 reduction has no ComputeReductionResult

---
 .../Transforms/Vectorize/LoopVectorize.cpp    | 23 +----
 .../Transforms/Vectorize/VPlanTransforms.cpp  | 43 ++++++---
 ...pilog-vectorization-anyof-no-rdx-result.ll | 93 +++++++++++++++++--
 .../LoopVectorize/select-cmp-blend-chain.ll   | 33 -------
 4 files changed, 117 insertions(+), 75 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 61813bd7bc8c6..de385ae0a4511 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -7619,32 +7619,19 @@ static SmallVector<Instruction *> preparePlanForEpilogueVectorLoop(
   SmallVector<Instruction *> InstsToMove;
   // Ensure that the start values for all header phi recipes are updated before
   // vectorizing the epilogue loop.
-  for (VPRecipeBase &R : make_early_inc_range(Header->phis())) {
+  for (VPRecipeBase &R : Header->phis()) {
     Value *ResumeV = nullptr;
     // TODO: Move setting of resume values to prepareToExecute.
     if (auto *ReductionPhi = dyn_cast<VPReductionPHIRecipe>(&R)) {
       // Find the reduction result by searching users of the phi or its backedge
-      // value, looking through intermediate recipes.
+      // value.
       auto IsReductionResult = [](VPRecipeBase *R) {
         auto *VPI = dyn_cast<VPInstruction>(R);
         return VPI && VPI->getOpcode() == VPInstruction::ComputeReductionResult;
       };
-      VPValue *BackedgeVal = ReductionPhi->getBackedgeValue();
-      auto *RdxResult = cast_or_null<VPInstruction>(
-          vputils::findRecipe(BackedgeVal, IsReductionResult));
-      // If the ComputeReductionResult was optimized away (e.g., the exit value
-      // was simplified to the start value), the reduction does not contribute
-      // to the exit value. Get rid of the dead reduction cycle.
-      if (!RdxResult) {
-        ReductionPhi->replaceAllUsesWith(ReductionPhi->getStartValue());
-        // The backedge value may be the phi itself (if the backedge chain was
-        // simplified away), or a separate recipe forming a cycle with the phi.
-        bool BackedgeIsPhi = (BackedgeVal == ReductionPhi);
-        ReductionPhi->eraseFromParent();
-        if (!BackedgeIsPhi)
-          vputils::recursivelyDeleteDeadRecipes(BackedgeVal);
-        continue;
-      }
+      auto *RdxResult = cast<VPInstruction>(
+          vputils::findRecipe(ReductionPhi->getBackedgeValue(), IsReductionResult));
+      assert(RdxResult && "expected to find reduction result");
 
       VPInstruction *ResumeForEpi = IRPhiToResumeForEpi.at(
           cast<PHINode>(ReductionPhi->getUnderlyingInstr()));
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 83e23449df5f5..9451d7ba71de1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -680,6 +680,32 @@ static void removeRedundantInductionCasts(VPlan &Plan) {
   }
 }
 
+/// If R is a phi-like recipe starting a dead cycle of recipes, erase all
+/// reachable recipes of the dead cycle.
+static void tryToRemoveDeadCycle(VPRecipeBase *R) {
+  auto *PhiR = dyn_cast<VPSingleDefRecipe>(R);
+  if (!PhiR || !isa<VPPhiAccessors>(R) || isa<VPCurrentIterationPHIRecipe>(R))
+    return;
+
+  // The transitive users of PhiR are closed under users, so the cycle is dead
+  // if every one of them can be erased.
+  for (VPUser *U : vputils::collectUsersRecursively(PhiR)) {
+    auto *R = cast<VPRecipeBase>(U);
+    // Bail out if a user must be retained, or if it is a phi-like recipe other
+    // than PhiR;
+    if (R->mayHaveSideEffects() || (R != PhiR && isa<VPPhiAccessors>(R)))
+      return;
+  }
+
+  // Break the cycle by replacing PhiR with its first incoming value, which is
+  // defined outside the cycle. That leaves the rest of the cycle dead.
+  PhiR->replaceAllUsesWith(PhiR->getOperand(0));
+  SmallVector<VPValue *> Incoming(PhiR->operands());
+  PhiR->eraseFromParent();
+  for (VPValue *Op : Incoming)
+    vputils::recursivelyDeleteDeadRecipes(Op);
+}
+
 void VPlanTransforms::removeDeadRecipes(VPlan &Plan) {
   PostOrderTraversal<VPBlockDeepTraversalWrapper<VPBlockBase *>> POT(
       Plan.getEntry());
@@ -692,20 +718,9 @@ void VPlanTransforms::removeDeadRecipes(VPlan &Plan) {
         continue;
       }
 
-      // Check if R is a dead VPPhi <-> update cycle and remove it.
-      VPValue *Start, *Incoming;
-      if (!match(&R, m_VPPhi(m_VPValue(Start), m_VPValue(Incoming))))
-        continue;
-      auto *PhiR = cast<VPPhi>(&R);
-      VPUser *PhiUser = PhiR->getSingleUser();
-      if (!PhiUser)
-        continue;
-      if (PhiUser != Incoming->getDefiningRecipe() ||
-          Incoming->getNumUsers() != 1)
-        continue;
-      PhiR->replaceAllUsesWith(Start);
-      PhiR->eraseFromParent();
-      Incoming->getDefiningRecipe()->eraseFromParent();
+      // If R is a phi-like recipe starting a dead cycle of recipes, erase the
+      // whole cycle.
+      tryToRemoveDeadCycle(&R);
     }
   }
 }
diff --git a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
index f80dce8cbafc5..8c74af2508c9f 100644
--- a/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
+++ b/llvm/test/Transforms/LoopVectorize/epilog-vectorization-anyof-no-rdx-result.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals none --version 6
 ; RUN: opt -passes=loop-vectorize -force-vector-width=4 \
 ; RUN:   -enable-epilogue-vectorization -epilogue-vectorization-force-VF=2 -S %s \
 ; RUN:   | FileCheck %s
@@ -17,7 +17,6 @@ define i8 @dead_anyof_reduction() {
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[VEC_PHI]], %[[VECTOR_BODY]] ]
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
 ; CHECK-NEXT:    [[TMP0:%.*]] = icmp eq i32 [[INDEX_NEXT]], -4
 ; CHECK-NEXT:    br i1 [[TMP0]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
@@ -66,11 +65,85 @@ loop:
   %exit.cond = icmp eq i32 %iv.next, 0
   br i1 %exit.cond, label %exit, label %loop
 }
-;.
-; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
-; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
-; CHECK: [[META2]] = !{!"llvm.loop.unroll.runtime.disable"}
-; CHECK: [[PROF3]] = !{!"branch_weights", i32 2, i32 2}
-; CHECK: [[LOOP4]] = distinct !{[[LOOP4]], [[META1]], [[META2]]}
-; CHECK: [[LOOP5]] = distinct !{[[LOOP5]], [[META2]], [[META1]]}
-;.
+
+define i32 @dead_anyof_reduction_blend(i1 %c0, i32 %n) {
+; CHECK-LABEL: define i32 @dead_anyof_reduction_blend(
+; CHECK-SAME: i1 [[C0:%.*]], i32 [[N:%.*]]) {
+; CHECK-NEXT:  [[ITER_CHECK:.*]]:
+; CHECK-NEXT:    [[SMAX:%.*]] = call i32 @llvm.smax.i32(i32 [[N]], i32 1)
+; CHECK-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i32 [[SMAX]], 2
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH:.*]], label %[[VECTOR_MAIN_LOOP_ITER_CHECK:.*]]
+; CHECK:       [[VECTOR_MAIN_LOOP_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_ITERS_CHECK1:%.*]] = icmp ult i32 [[SMAX]], 4
+; CHECK-NEXT:    br i1 [[MIN_ITERS_CHECK1]], label %[[VEC_EPILOG_PH:.*]], label %[[VECTOR_PH:.*]]
+; CHECK:       [[VECTOR_PH]]:
+; CHECK-NEXT:    [[TMP0:%.*]] = urem i32 [[SMAX]], 4
+; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[SMAX]], [[TMP0]]
+; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
+; CHECK:       [[VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT:    [[TMP1:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[TMP1]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK:       [[MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N:%.*]] = icmp eq i32 [[SMAX]], [[N_VEC]]
+; CHECK-NEXT:    br i1 [[CMP_N]], label %[[EXIT:.*]], label %[[VEC_EPILOG_ITER_CHECK:.*]]
+; CHECK:       [[VEC_EPILOG_ITER_CHECK]]:
+; CHECK-NEXT:    [[MIN_EPILOG_ITERS_CHECK:%.*]] = icmp ult i32 [[TMP0]], 2
+; CHECK-NEXT:    br i1 [[MIN_EPILOG_ITERS_CHECK]], label %[[VEC_EPILOG_SCALAR_PH]], label %[[VEC_EPILOG_PH]], !prof [[PROF3]]
+; CHECK:       [[VEC_EPILOG_PH]]:
+; CHECK-NEXT:    [[VEC_EPILOG_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[VECTOR_MAIN_LOOP_ITER_CHECK]] ]
+; CHECK-NEXT:    [[TMP2:%.*]] = urem i32 [[SMAX]], 2
+; CHECK-NEXT:    [[N_VEC2:%.*]] = sub i32 [[SMAX]], [[TMP2]]
+; CHECK-NEXT:    br label %[[VEC_EPILOG_VECTOR_BODY:.*]]
+; CHECK:       [[VEC_EPILOG_VECTOR_BODY]]:
+; CHECK-NEXT:    [[INDEX3:%.*]] = phi i32 [ [[VEC_EPILOG_RESUME_VAL]], %[[VEC_EPILOG_PH]] ], [ [[INDEX_NEXT4:%.*]], %[[VEC_EPILOG_VECTOR_BODY]] ]
+; CHECK-NEXT:    [[INDEX_NEXT4]] = add nuw i32 [[INDEX3]], 2
+; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT4]], [[N_VEC2]]
+; CHECK-NEXT:    br i1 [[TMP3]], label %[[VEC_EPILOG_MIDDLE_BLOCK:.*]], label %[[VEC_EPILOG_VECTOR_BODY]], !llvm.loop [[LOOP7:![0-9]+]]
+; CHECK:       [[VEC_EPILOG_MIDDLE_BLOCK]]:
+; CHECK-NEXT:    [[CMP_N5:%.*]] = icmp eq i32 [[SMAX]], [[N_VEC2]]
+; CHECK-NEXT:    br i1 [[CMP_N5]], label %[[EXIT]], label %[[VEC_EPILOG_SCALAR_PH]]
+; CHECK:       [[VEC_EPILOG_SCALAR_PH]]:
+; CHECK-NEXT:    [[BC_RESUME_VAL:%.*]] = phi i32 [ [[N_VEC2]], %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ [[N_VEC]], %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT:    [[BC_MERGE_RDX7:%.*]] = phi i32 [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_ITER_CHECK]] ], [ 0, %[[ITER_CHECK]] ]
+; CHECK-NEXT:    br label %[[LOOP_HEADER:.*]]
+; CHECK:       [[LOOP_HEADER]]:
+; CHECK-NEXT:    [[IV:%.*]] = phi i32 [ [[BC_RESUME_VAL]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[IV_NEXT:%.*]], %[[LOOP_LATCH:.*]] ]
+; CHECK-NEXT:    [[RDX:%.*]] = phi i32 [ [[BC_MERGE_RDX7]], %[[VEC_EPILOG_SCALAR_PH]] ], [ [[RDX_NEXT:%.*]], %[[LOOP_LATCH]] ]
+; CHECK-NEXT:    br i1 [[C0]], label %[[LOOP_LATCH]], label %[[IF_THEN:.*]]
+; CHECK:       [[IF_THEN]]:
+; CHECK-NEXT:    [[C:%.*]] = icmp eq i32 [[IV]], 0
+; CHECK-NEXT:    [[SEL:%.*]] = select i1 [[C]], i32 0, i32 [[RDX]]
+; CHECK-NEXT:    br label %[[LOOP_LATCH]]
+; CHECK:       [[LOOP_LATCH]]:
+; CHECK-NEXT:    [[RDX_NEXT]] = phi i32 [ [[SEL]], %[[IF_THEN]] ], [ [[RDX]], %[[LOOP_HEADER]] ]
+; CHECK-NEXT:    [[IV_NEXT]] = add i32 [[IV]], 1
+; CHECK-NEXT:    [[EC:%.*]] = icmp slt i32 [[IV_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[EC]], label %[[LOOP_HEADER]], label %[[EXIT]], !llvm.loop [[LOOP8:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    [[RDX_NEXT_LCSSA:%.*]] = phi i32 [ [[RDX_NEXT]], %[[LOOP_LATCH]] ], [ 0, %[[MIDDLE_BLOCK]] ], [ 0, %[[VEC_EPILOG_MIDDLE_BLOCK]] ]
+; CHECK-NEXT:    ret i32 [[RDX_NEXT_LCSSA]]
+;
+entry:
+  br label %loop.header
+
+loop.header:
+  %iv = phi i32 [ 0, %entry ], [ %iv.next, %loop.latch ]
+  %rdx = phi i32 [ 0, %entry ], [ %rdx.next, %loop.latch ]
+  br i1 %c0, label %loop.latch, label %if.then
+
+if.then:
+  %c = icmp eq i32 %iv, 0
+  %sel = select i1 %c, i32 0, i32 %rdx
+  br label %loop.latch
+
+loop.latch:
+  %rdx.next = phi i32 [ %sel, %if.then ], [ %rdx, %loop.header ]
+  %iv.next = add i32 %iv, 1
+  %ec = icmp slt i32 %iv.next, %n
+  br i1 %ec, label %loop.header, label %exit
+
+exit:
+  ret i32 %rdx.next
+}
diff --git a/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll b/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
index 8b12e09631246..4318798ffadda 100644
--- a/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
+++ b/llvm/test/Transforms/LoopVectorize/select-cmp-blend-chain.ll
@@ -15,14 +15,7 @@ define i32 @anyof_two_blend_chain(i1 %c0, i1 %c1, i32 %n) {
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI1:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP0:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT:    [[TMP1:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP0]]
-; CHECK-NEXT:    [[PREDPHI:%.*]] = select i1 [[C1]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[PREDPHI1]] = select i1 [[C0]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI]]
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
 ; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP0:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
@@ -97,15 +90,7 @@ define i32 @anyof_three_blend_chain(i1 %c0, i1 %c1, i1 %c2, i32 %n) {
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI2:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP0:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT:    [[TMP1:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP0]]
-; CHECK-NEXT:    [[PREDPHI:%.*]] = select i1 [[C2]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP1]]
-; CHECK-NEXT:    [[PREDPHI1:%.*]] = select i1 [[C1]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI]]
-; CHECK-NEXT:    [[PREDPHI2]] = select i1 [[C0]], <4 x i1> [[VEC_PHI]], <4 x i1> [[PREDPHI1]]
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
 ; CHECK-NEXT:    [[TMP2:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP2]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
@@ -189,21 +174,10 @@ define i32 @anyof_diamond_blend_chain(i1 %c0, i1 %c1, i32 %n) {
 ; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i32 [[SMAX]], 4
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[SMAX]], [[N_MOD_VF]]
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i1> poison, i1 [[C1]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT1]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[BROADCAST_SPLATINSERT2:%.*]] = insertelement <4 x i1> poison, i1 [[C0]], i64 0
-; CHECK-NEXT:    [[BROADCAST_SPLAT3:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT2]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT:    [[TMP1:%.*]] = select <4 x i1> [[BROADCAST_SPLAT2]], <4 x i1> [[BROADCAST_SPLAT3]], <4 x i1> zeroinitializer
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[TMP3:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP0:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP0]], <4 x i1> zeroinitializer
-; CHECK-NEXT:    [[TMP3]] = or <4 x i1> [[VEC_PHI]], [[TMP2]]
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
 ; CHECK-NEXT:    [[TMP4:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:
@@ -293,17 +267,10 @@ define i32 @anyof_blend_with_select_mask(i1 %a, i1 %b, i32 %n) {
 ; CHECK:       [[VECTOR_PH]]:
 ; CHECK-NEXT:    [[N_MOD_VF:%.*]] = urem i32 [[SMAX]], 4
 ; CHECK-NEXT:    [[N_VEC:%.*]] = sub i32 [[SMAX]], [[N_MOD_VF]]
-; CHECK-NEXT:    [[TMP0:%.*]] = select i1 [[A]], i1 [[B]], i1 false
 ; CHECK-NEXT:    br label %[[VECTOR_BODY:.*]]
 ; CHECK:       [[VECTOR_BODY]]:
 ; CHECK-NEXT:    [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_IND:%.*]] = phi <4 x i32> [ <i32 0, i32 1, i32 2, i32 3>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[VEC_PHI:%.*]] = phi <4 x i1> [ zeroinitializer, %[[VECTOR_PH]] ], [ [[PREDPHI:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT:    [[TMP1:%.*]] = icmp eq <4 x i32> [[VEC_IND]], zeroinitializer
-; CHECK-NEXT:    [[TMP2:%.*]] = or <4 x i1> [[VEC_PHI]], [[TMP1]]
-; CHECK-NEXT:    [[PREDPHI]] = select i1 [[TMP0]], <4 x i1> [[VEC_PHI]], <4 x i1> [[TMP2]]
 ; CHECK-NEXT:    [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
-; CHECK-NEXT:    [[VEC_IND_NEXT]] = add <4 x i32> [[VEC_IND]], splat (i32 4)
 ; CHECK-NEXT:    [[TMP3:%.*]] = icmp eq i32 [[INDEX_NEXT]], [[N_VEC]]
 ; CHECK-NEXT:    br i1 [[TMP3]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP8:![0-9]+]]
 ; CHECK:       [[MIDDLE_BLOCK]]:



More information about the llvm-commits mailing list