[llvm] [VPlan] Use CFG to mask early exit loops with side-effects. NFC (PR #203263)
Luke Lau via llvm-commits
llvm-commits at lists.llvm.org
Thu Jul 23 02:32:27 PDT 2026
https://github.com/lukel97 updated https://github.com/llvm/llvm-project/pull/203263
>From cc952159d885eeb4547220fb29eaeaf9fe1f4d06 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Tue, 9 Jun 2026 20:10:59 +0800
Subject: [PATCH 1/9] Precommit tests
---
.../LoopVectorize/VPlan/predicator.ll | 79 +++++++++++++++++++
.../Transforms/LoopVectorize/predicator.ll | 72 +++++++++++++++++
2 files changed, 151 insertions(+)
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index 483ab8ad94b50..d9766bbc5c4c7 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -959,3 +959,82 @@ latch:
exit:
ret void
}
+
+; loop
+; / \
+; A \
+; / \ B
+; \ \ /
+; C D
+; \/
+; latch
+define void @look_thru_phi(i1 %c1, i1 %c2, i32 %x, i32 %y, ptr %p) {
+; CHECK-LABEL: VPlan for loop in 'look_thru_phi'
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK-NEXT: Successor(s): B
+; CHECK-EMPTY:
+; CHECK-NEXT: B:
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = not ir<%c1>
+; CHECK-NEXT: Successor(s): A
+; CHECK-EMPTY:
+; CHECK-NEXT: A:
+; CHECK-NEXT: Successor(s): D
+; CHECK-EMPTY:
+; CHECK-NEXT: D:
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = not ir<%c2>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and ir<%c1>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = or vp<[[VP4]]>, vp<[[VP6]]>
+; CHECK-NEXT: BLEND ir<%phi1> = ir<%y>/vp<[[VP4]]> ir<%x>/vp<[[VP6]]>
+; CHECK-NEXT: Successor(s): C
+; CHECK-EMPTY:
+; CHECK-NEXT: C:
+; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = logical-and ir<%c1>, ir<%c2>
+; CHECK-NEXT: Successor(s): latch
+; CHECK-EMPTY:
+; CHECK-NEXT: latch:
+; CHECK-NEXT: BLEND ir<%phi> = ir<%phi1>/vp<[[VP7]]> ir<%x>/vp<[[VP8]]>
+; CHECK-NEXT: EMIT ir<%gep> = getelementptr ir<%p>, ir<%iv>
+; CHECK-NEXT: EMIT store ir<%phi>, ir<%gep>
+; CHECK-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; CHECK-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<128>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK-NEXT: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [0, %entry], [%iv.next, %latch]
+ br i1 %c1, label %A, label %B
+
+A:
+ br i1 %c2, label %C, label %D
+
+B:
+ br label %D
+
+C:
+ br label %latch
+
+D:
+ %phi1 = phi i32 [%x, %A], [%y, %B]
+ br label %latch
+
+latch:
+ %phi = phi i32 [ %x, %C ], [ %phi1, %D ]
+ %gep = getelementptr i32, ptr %p, i32 %iv
+ store i32 %phi, ptr %gep
+ %iv.next = add i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, 128
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/predicator.ll b/llvm/test/Transforms/LoopVectorize/predicator.ll
index 57414ae62c341..a03855edd8de7 100644
--- a/llvm/test/Transforms/LoopVectorize/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicator.ll
@@ -292,3 +292,75 @@ latch:
exit:
ret void
}
+
+; loop
+; / \
+; A \
+; / \ B
+; \ \ /
+; C D
+; \/
+; latch
+define void @look_thru_phi(i1 %c1, i1 %c2, i32 %x, i32 %y, ptr %p) {
+; CHECK-LABEL: define void @look_thru_phi(
+; CHECK-SAME: i1 [[C1:%.*]], i1 [[C2:%.*]], i32 [[X:%.*]], i32 [[Y:%.*]], ptr [[P:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[Y]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT1]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <4 x i1> poison, i1 [[C2]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT4:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT3]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT5:%.*]] = insertelement <4 x i1> poison, i1 [[C1]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT6:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT5]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP0:%.*]] = xor <4 x i1> [[BROADCAST_SPLAT4]], splat (i1 true)
+; CHECK-NEXT: [[TMP1:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[TMP0]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP1]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[TMP2:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[BROADCAST_SPLAT4]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[PREDPHI7:%.*]] = select <4 x i1> [[TMP2]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[PREDPHI]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i32, ptr [[P]], i32 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[PREDPHI7]], ptr [[TMP3]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i32 [[INDEX_NEXT]], 128
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [0, %entry], [%iv.next, %latch]
+ br i1 %c1, label %A, label %B
+
+A:
+ br i1 %c2, label %C, label %D
+
+B:
+ br label %D
+
+C:
+ br label %latch
+
+D:
+ %phi1 = phi i32 [%x, %A], [%y, %B]
+ br label %latch
+
+latch:
+ %phi = phi i32 [ %x, %C ], [ %phi1, %D ]
+ %gep = getelementptr i32, ptr %p, i32 %iv
+ store i32 %phi, ptr %gep
+ %iv.next = add i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, 128
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
>From 3f83cc01a249ef2567ceb41e03d9aad41f640db0 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Tue, 9 Jun 2026 20:12:38 +0800
Subject: [PATCH 2/9] Peek through inner phi's nested values
---
.../Transforms/Vectorize/VPlanPredicator.cpp | 21 +++++++++++++++++++
.../LoopVectorize/VPlan/predicator.ll | 10 +++------
.../Transforms/LoopVectorize/predicator.ll | 10 +--------
3 files changed, 25 insertions(+), 16 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 3f35f14e876f2..9a15dd853b8bf 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -275,6 +275,10 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
for (auto [InVal, InVPBB] : Phi->incoming_values_and_blocks())
AddEdge(InVPBB, Phi->getParent(), InVal);
+ // Don't optimize any reduction chains for now.
+ if (any_of(Phi->incoming_values(), IsaPred<VPReductionPHIRecipe>))
+ return Edges;
+
SetVector<const VPBlockBase *> Worklist(from_range, Phi->incoming_blocks());
while (!Worklist.empty()) {
auto *VPBB = cast<VPBasicBlock>(Worklist.pop_back_val());
@@ -293,6 +297,17 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
for (EdgeTy Edge : OutEdges)
Edges.erase(Edge);
+ // If the value is a phi postdominated by VPBB, then look through the inner
+ // incoming values instead of propagating the phi.
+ if (auto *Phi = dyn_cast<VPPhi>(Common))
+ if (VPPDT.dominates(VPBB, Phi->getParent())) {
+ for (auto [InV, InVPBB] : Phi->incoming_values_and_blocks()) {
+ AddEdge(InVPBB, Phi->getParent(), InV);
+ Worklist.insert(InVPBB);
+ }
+ continue;
+ }
+
// Iterate up through the post dominance frontier.
assert(VPPDF.find(VPBB) != VPPDF.end() &&
"VPBB must have a post-dominance frontier entry");
@@ -373,6 +388,12 @@ void VPPredicator::convertPhisToBlends(VPBasicBlock *VPBB) {
InValEdgesMap[Val].push_back(Edge);
auto InValEdges = InValEdgesMap.takeVector();
+ if (InValEdges.size() == 1) {
+ PhiR->replaceAllUsesWith(InValEdges[0].first);
+ PhiR->eraseFromParent();
+ continue;
+ }
+
// Sort the incoming value order to match PhiR as much as possible.
llvm::stable_sort(InValEdges, [&PhiR](auto &L, auto &R) {
auto InVs = PhiR->incoming_values();
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index d9766bbc5c4c7..84f681958550f 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -665,8 +665,6 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): merge.a
; CHECK-EMPTY:
; CHECK-NEXT: merge.a:
-; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = not ir<%c0>
-; CHECK-NEXT: BLEND ir<%blend.a> = ir<%v1>/ir<%c0> ir<%v1>/vp<[[VP5]]>
; CHECK-NEXT: EMIT ir<%d0> = icmp sgt ir<%iv>, ir<0>
; CHECK-NEXT: Successor(s): if.b
; CHECK-EMPTY:
@@ -675,16 +673,14 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): if.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: if.b.inner:
-; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
; CHECK-NEXT: Successor(s): merge.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: merge.b.inner:
; CHECK-NEXT: Successor(s): merge.b
; CHECK-EMPTY:
; CHECK-NEXT: merge.b:
-; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = not ir<%d0>
-; CHECK-NEXT: BLEND ir<%blend.b> = ir<%v2>/ir<%d0> ir<%v2>/vp<[[VP7]]>
-; CHECK-NEXT: EMIT ir<%sum> = add ir<%blend.a>, ir<%blend.b>
+; CHECK-NEXT: EMIT ir<%sum> = add ir<%v1>, ir<%v2>
; CHECK-NEXT: EMIT store ir<%sum>, ir<%gep>
; CHECK-NEXT: Successor(s): loop.latch
; CHECK-EMPTY:
@@ -996,7 +992,7 @@ define void @look_thru_phi(i1 %c1, i1 %c2, i32 %x, i32 %y, ptr %p) {
; CHECK-NEXT: Successor(s): latch
; CHECK-EMPTY:
; CHECK-NEXT: latch:
-; CHECK-NEXT: BLEND ir<%phi> = ir<%phi1>/vp<[[VP7]]> ir<%x>/vp<[[VP8]]>
+; CHECK-NEXT: BLEND ir<%phi> = ir<%x>/ir<%c1> ir<%y>/vp<[[VP4]]>
; CHECK-NEXT: EMIT ir<%gep> = getelementptr ir<%p>, ir<%iv>
; CHECK-NEXT: EMIT store ir<%phi>, ir<%gep>
; CHECK-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
diff --git a/llvm/test/Transforms/LoopVectorize/predicator.ll b/llvm/test/Transforms/LoopVectorize/predicator.ll
index a03855edd8de7..98529cb1c227d 100644
--- a/llvm/test/Transforms/LoopVectorize/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicator.ll
@@ -311,15 +311,7 @@ define void @look_thru_phi(i1 %c1, i1 %c2, i32 %x, i32 %y, ptr %p) {
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT1]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <4 x i1> poison, i1 [[C2]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT4:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT3]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT5:%.*]] = insertelement <4 x i1> poison, i1 [[C1]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT6:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT5]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP0:%.*]] = xor <4 x i1> [[BROADCAST_SPLAT4]], splat (i1 true)
-; CHECK-NEXT: [[TMP1:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[TMP0]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP1]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[BROADCAST_SPLAT]]
-; CHECK-NEXT: [[TMP2:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[BROADCAST_SPLAT4]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[PREDPHI7:%.*]] = select <4 x i1> [[TMP2]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[PREDPHI]]
+; CHECK-NEXT: [[PREDPHI7:%.*]] = select i1 [[C1]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
>From bc80b98783a6a5ffe24a921e0e0ab1b0fb9f646a Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Fri, 19 Jun 2026 13:33:57 +0200
Subject: [PATCH 3/9] Don't peek thru phis with multiple uses
---
.../Transforms/Vectorize/VPlanPredicator.cpp | 2 +-
.../LoopVectorize/VPlan/predicator.ll | 76 +++++++++++++++++++
2 files changed, 77 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 9a15dd853b8bf..89e7ec74f5e3d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -300,7 +300,7 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
// If the value is a phi postdominated by VPBB, then look through the inner
// incoming values instead of propagating the phi.
if (auto *Phi = dyn_cast<VPPhi>(Common))
- if (VPPDT.dominates(VPBB, Phi->getParent())) {
+ if (Phi->hasOneUse() && VPPDT.dominates(VPBB, Phi->getParent())) {
for (auto [InV, InVPBB] : Phi->incoming_values_and_blocks()) {
AddEdge(InVPBB, Phi->getParent(), InV);
Worklist.insert(InVPBB);
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index 84f681958550f..6359e2e0d4200 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -1034,3 +1034,79 @@ latch:
exit:
ret void
}
+
+
+; Same as the test above, but shouldn't look thru the phi because it has
+; multiple uses.
+define void @look_thru_phi_multi_use(i1 %c1, i1 %c2, i32 %x, i32 %y, ptr %p) {
+; CHECK-LABEL: VPlan for loop in 'look_thru_phi_multi_use'
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK-NEXT: EMIT ir<%gep> = getelementptr ir<%p>, ir<%iv>
+; CHECK-NEXT: Successor(s): B
+; CHECK-EMPTY:
+; CHECK-NEXT: B:
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = not ir<%c1>
+; CHECK-NEXT: Successor(s): A
+; CHECK-EMPTY:
+; CHECK-NEXT: A:
+; CHECK-NEXT: Successor(s): D
+; CHECK-EMPTY:
+; CHECK-NEXT: D:
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = not ir<%c2>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and ir<%c1>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = or vp<[[VP4]]>, vp<[[VP6]]>
+; CHECK-NEXT: BLEND ir<%phi1> = ir<%y>/vp<[[VP4]]> ir<%x>/vp<[[VP6]]>
+; CHECK-NEXT: EMIT store ir<%phi1>, ir<%gep>, vp<[[VP7]]>
+; CHECK-NEXT: Successor(s): C
+; CHECK-EMPTY:
+; CHECK-NEXT: C:
+; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = logical-and ir<%c1>, ir<%c2>
+; CHECK-NEXT: Successor(s): latch
+; CHECK-EMPTY:
+; CHECK-NEXT: latch:
+; CHECK-NEXT: BLEND ir<%phi> = ir<%phi1>/vp<[[VP7]]> ir<%x>/vp<[[VP8]]>
+; CHECK-NEXT: EMIT store ir<%phi>, ir<%gep>
+; CHECK-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; CHECK-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<128>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK-NEXT: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [0, %entry], [%iv.next, %latch]
+ %gep = getelementptr i32, ptr %p, i32 %iv
+ br i1 %c1, label %A, label %B
+
+A:
+ br i1 %c2, label %C, label %D
+
+B:
+ br label %D
+
+C:
+ br label %latch
+
+D:
+ %phi1 = phi i32 [%x, %A], [%y, %B]
+ store i32 %phi1, ptr %gep
+ br label %latch
+
+latch:
+ %phi = phi i32 [ %x, %C ], [ %phi1, %D ]
+ store i32 %phi, ptr %gep
+ %iv.next = add i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, 128
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
>From 57d91a2dcf8ff8bad8a33cb3b0fcd6039a080f3e Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Fri, 19 Jun 2026 13:59:59 +0200
Subject: [PATCH 4/9] Move diagram below check lines
---
.../Transforms/LoopVectorize/VPlan/predicator.ll | 16 ++++++++--------
1 file changed, 8 insertions(+), 8 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index 6359e2e0d4200..37f03b102c6e4 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -956,14 +956,6 @@ exit:
ret void
}
-; loop
-; / \
-; A \
-; / \ B
-; \ \ /
-; C D
-; \/
-; latch
define void @look_thru_phi(i1 %c1, i1 %c2, i32 %x, i32 %y, ptr %p) {
; CHECK-LABEL: VPlan for loop in 'look_thru_phi'
; CHECK-NEXT: <x1> vector loop: {
@@ -1003,6 +995,14 @@ define void @look_thru_phi(i1 %c1, i1 %c2, i32 %x, i32 %y, ptr %p) {
; CHECK-NEXT: }
; CHECK-NEXT: Successor(s): middle.block
;
+; loop
+; / \
+; A \
+; / \ B
+; \ \ /
+; C D
+; \/
+; latch
entry:
br label %loop
>From dfc51d57deecc4b64fba98be7ff7d7f288a72ab0 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Mon, 13 Jul 2026 17:26:06 +0800
Subject: [PATCH 5/9] Update uniform-blend.ll
---
.../Transforms/LoopVectorize/uniform-blend.ll | 15 +++------------
1 file changed, 3 insertions(+), 12 deletions(-)
diff --git a/llvm/test/Transforms/LoopVectorize/uniform-blend.ll b/llvm/test/Transforms/LoopVectorize/uniform-blend.ll
index 8d86c8ca2bd9d..a016a26d479ec 100644
--- a/llvm/test/Transforms/LoopVectorize/uniform-blend.ll
+++ b/llvm/test/Transforms/LoopVectorize/uniform-blend.ll
@@ -96,19 +96,10 @@ define void @blend_chain_iv(i1 %c) {
; CHECK: [[VECTOR_PH]]:
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
-; CHECK-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
-; CHECK-NEXT: [[TMP3:%.*]] = add i64 [[INDEX]], 1
-; CHECK-NEXT: [[TMP5:%.*]] = add i64 [[INDEX]], 2
-; CHECK-NEXT: [[TMP7:%.*]] = add i64 [[INDEX]], 3
-; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds [32 x i16], ptr @dst, i16 0, i64 [[INDEX]]
-; CHECK-NEXT: [[TMP4:%.*]] = getelementptr inbounds [32 x i16], ptr @dst, i16 0, i64 [[TMP3]]
-; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds [32 x i16], ptr @dst, i16 0, i64 [[TMP5]]
+; CHECK-NEXT: [[TMP7:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds [32 x i16], ptr @dst, i16 0, i64 [[TMP7]]
-; CHECK-NEXT: store i16 0, ptr [[TMP2]], align 2
-; CHECK-NEXT: store i16 0, ptr [[TMP4]], align 2
-; CHECK-NEXT: store i16 0, ptr [[TMP6]], align 2
-; CHECK-NEXT: store i16 0, ptr [[TMP8]], align 2
-; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
+; CHECK-NEXT: store <4 x i16> zeroinitializer, ptr [[TMP8]], align 2
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[TMP7]], 4
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq i64 [[INDEX_NEXT]], 32
; CHECK-NEXT: br i1 [[TMP9]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP4:![0-9]+]]
; CHECK: [[MIDDLE_BLOCK]]:
>From fb5544a83c6e2cdd9949365646209f79bd7ac042 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Thu, 4 Jun 2026 13:10:07 +0800
Subject: [PATCH 6/9] Maintain SSA by replacing MaskedCond with phis
---
llvm/lib/Transforms/Vectorize/VPlan.h | 4 -
.../Transforms/Vectorize/VPlanLowering.cpp | 11 ---
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 8 --
.../Transforms/Vectorize/VPlanTransforms.cpp | 92 +++++++++++++++----
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 1 -
.../Transforms/Vectorize/VPlanVerifier.cpp | 18 ----
.../predicated-early-exits-interleave.ll | 18 ++--
.../predicated-multiple-exits.ll | 38 ++++----
.../LoopVectorize/single_early_exit.ll | 62 +++++++++++++
9 files changed, 162 insertions(+), 90 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index edb24141f6073..d4a6f9b664822 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1328,7 +1328,6 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
/// is the value of the last lane of the induction increment (i.e. its
/// backedge value). Has the wide induction recipe as operand.
ExitingIVValue,
- MaskedCond,
// The opcodes below are used for VPInstructionWithType.
// NOTE: VPInstructionWithType classes are also used for:
@@ -1374,9 +1373,6 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
/// Returns true if the VPInstruction does not need masking.
bool alwaysUnmasked() const {
- if (Opcode == VPInstruction::MaskedCond)
- return false;
-
// For now only VPInstructions with underlying values use masks.
// TODO: provide masks to VPInstructions w/o underlying values.
if (!getUnderlyingValue())
diff --git a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
index c92a551d5805d..550825b0a49e3 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanLowering.cpp
@@ -632,17 +632,6 @@ void VPlanTransforms::convertToConcreteRecipes(VPlan &Plan) {
continue;
}
- // Lower MaskedCond with block mask to LogicalAnd.
- if (match(&R, m_VPInstruction<VPInstruction::MaskedCond>())) {
- auto *VPI = cast<VPInstruction>(&R);
- assert(VPI->isMasked() &&
- "Unmasked MaskedCond should be simplified earlier");
- VPI->replaceAllUsesWith(Builder.createNaryOp(
- VPInstruction::LogicalAnd, {VPI->getMask(), VPI->getOperand(0)}));
- VPI->eraseFromParent();
- continue;
- }
-
// Lower CanonicalIVIncrementForPart to plain Add.
if (match(
&R,
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 6472d7f84b712..a0b7824dad559 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -500,9 +500,6 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
assert(Op0Ty->isIntegerTy() && "expected integer operand");
AssertOperandType(1, Op0Ty);
return IntegerType::get(Ctx, 1);
- case VPInstruction::MaskedCond:
- assert(Op0Ty->isIntegerTy(1) && "expected bool operand");
- return IntegerType::get(Ctx, 1);
case VPInstruction::LogicalAnd:
case VPInstruction::LogicalOr:
assert(Op0Ty->isIntegerTy(1) && "expected bool operand");
@@ -638,7 +635,6 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
case VPInstruction::ExtractLastLane:
case VPInstruction::ExtractLastPart:
case VPInstruction::ExtractPenultimateElement:
- case VPInstruction::MaskedCond:
case VPInstruction::Not:
case VPInstruction::Reverse:
case VPInstruction::Unpack:
@@ -1631,7 +1627,6 @@ bool VPInstruction::opcodeMayReadOrWriteFromMemory() const {
case VPInstruction::FirstOrderRecurrenceSplice:
case VPInstruction::LogicalAnd:
case VPInstruction::LogicalOr:
- case VPInstruction::MaskedCond:
case VPInstruction::Not:
case VPInstruction::PtrAdd:
case VPInstruction::WideIVStep:
@@ -1786,9 +1781,6 @@ void VPInstruction::printRecipe(raw_ostream &O, const Twine &Indent,
case VPInstruction::ExitingIVValue:
O << "exiting-iv-value";
break;
- case VPInstruction::MaskedCond:
- O << "masked-cond";
- break;
case VPInstruction::ExtractLane:
O << "extract-lane";
break;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 23d7db861e6aa..c4b55e029a598 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -1443,11 +1443,6 @@ static void simplifyRecipe(VPSingleDefRecipe *Def) {
return;
}
- // Simplify MaskedCond with no block mask to its single operand.
- if (match(Def, m_VPInstruction<VPInstruction::MaskedCond>()) &&
- !cast<VPInstruction>(Def)->isMasked())
- return Def->replaceAllUsesWith(Def->getOperand(0));
-
// Look through ExtractLastLane.
if (match(Def, m_ExtractLastLane(m_VPValue(A)))) {
if (match(A, m_BuildVector())) {
@@ -2983,10 +2978,6 @@ getRecipesForUncountableExit(SmallVectorImpl<VPInstruction *> &Recipes,
return std::nullopt;
Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
Recipes.push_back(cast<VPInstruction>(GepR));
- } else if (match(V, m_VPInstruction<VPInstruction::MaskedCond>(
- m_VPValue(Op1)))) {
- Worklist.push_back(Op1);
- Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
} else
return std::nullopt;
}
@@ -3006,6 +2997,74 @@ struct EarlyExitInfo {
VPIRBasicBlock *EarlyExitVPBB;
VPValue *CondToExit;
};
+static VPValue *repairSSA(VPValue *Src, VPBasicBlock *SrcVPBB, VPValue *Other,
+ VPBasicBlock *VPBB, VPDominatorTree &VPDT,
+ DenseMap<VPBlockBase *, VPPhi *> &Phis) {
+
+ if (VPDT.dominates(SrcVPBB, VPBB))
+ return Src;
+ if (VPDT.dominates(VPBB, SrcVPBB))
+ return Other;
+ if (VPPhi *Phi = Phis.lookup(VPBB))
+ return Phi;
+
+ SmallVector<VPValue *> InVals;
+ for (auto *Pred : VPBB->predecessors())
+ InVals.push_back(
+ repairSSA(Src, SrcVPBB, Other, cast<VPBasicBlock>(Pred), VPDT, Phis));
+ if (all_equal(InVals))
+ return InVals[0];
+
+ VPPhi *Phi = VPBuilder(VPBB, VPBB->getFirstNonPhi()).createScalarPhi(InVals);
+ Phis[VPBB] = Phi;
+ return Phi;
+}
+
+/// Insert phi nodes to maintain SSA starting from \p VPBB, such that the
+/// resulting value is \p \Src on all paths that go through \p SrcVPBB, and \p
+/// Other otherwise.
+static VPValue *repairSSA(VPValue *Src, VPBasicBlock *SrcVPBB, VPValue *Other,
+ VPBasicBlock *VPBB, VPDominatorTree &VPDT) {
+ DenseMap<VPBlockBase *, VPPhi *> Phis;
+ return repairSSA(Src, SrcVPBB, Other, VPBB, VPDT, Phis);
+}
+
+// After handling early exits, the CondToExits and live outs may no longer be in
+// SSA if their defining blocks are predicated, so insert phis to repair them.
+static void repairEarlyExitSSA(VPlan &Plan, VPDominatorTree &VPDT,
+ ArrayRef<EarlyExitInfo> Exits,
+ VPBasicBlock *LatchVPBB,
+ ArrayRef<VPBasicBlock *> LiveOutVPBBs) {
+ // Repair all CondToExits. The condition is false on any path that doesn't go
+ // through the exiting block.
+ for (auto [EarlyExitingVPBB, _, CondToExit] : Exits) {
+ VPValue *Repaired = repairSSA(CondToExit, EarlyExitingVPBB, Plan.getFalse(),
+ LatchVPBB, VPDT);
+
+ CondToExit->replaceUsesWithIf(Repaired, [&](VPUser &U, unsigned I) {
+ auto &R = cast<VPRecipeBase>(U);
+ return VPDT.dominates(LatchVPBB, R.getParent()) &&
+ R.getVPSingleValue() != Repaired;
+ });
+ }
+
+ // Repair any live outs. The value is poison on any path that didn't pass
+ // through the def's block.
+ for (VPBasicBlock *LiveOutVPBB : LiveOutVPBBs)
+ for (VPRecipeBase &R : *LiveOutVPBB) {
+ VPValue *LiveOut;
+ if (!match(&R,
+ m_CombineOr(m_ExtractLastPart(m_VPValue(LiveOut)),
+ m_ExtractLane(m_VPValue(), m_VPValue(LiveOut)))))
+ continue;
+ VPValue *Poison =
+ Plan.getOrAddLiveIn(PoisonValue::get(LiveOut->getScalarType()));
+ VPValue *Repaired =
+ repairSSA(LiveOut, LiveOut->getDefiningRecipe()->getParent(), Poison,
+ LatchVPBB, VPDT);
+ R.replaceUsesOfWith(LiveOut, Repaired);
+ }
+}
/// Update \p Plan to mask memory operations in the loop based on whether the
/// early exit is taken or not.
@@ -3194,9 +3253,7 @@ bool VPlanTransforms::handleUncountableEarlyExits(
VPlan &Plan, VPBasicBlock *HeaderVPBB, VPBasicBlock *LatchVPBB,
VPBasicBlock *MiddleVPBB, Loop *TheLoop, PredicatedScalarEvolution &PSE,
DominatorTree &DT, AssumptionCache *AC, UncountableExitStyle Style) {
-#ifndef NDEBUG
VPDominatorTree VPDT(Plan);
-#endif
VPBuilder LatchBuilder(LatchVPBB->getTerminator());
SmallVector<EarlyExitInfo> Exits;
for (VPIRBasicBlock *ExitBlock : Plan.getExitBlocks()) {
@@ -3212,14 +3269,12 @@ bool VPlanTransforms::handleUncountableEarlyExits(
m_BranchOnCond(m_VPValue(CondOfEarlyExitingVPBB)));
assert(Matched && "Terminator must be BranchOnCond");
- // Insert the MaskedCond in the EarlyExitingVPBB so the predicator adds
- // the correct block mask.
VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
- auto *CondToEarlyExit = EarlyExitingBuilder.createNaryOp(
- VPInstruction::MaskedCond,
+ auto *CondToEarlyExit =
TrueSucc == ExitBlock
? CondOfEarlyExitingVPBB
- : EarlyExitingBuilder.createNot(CondOfEarlyExitingVPBB));
+ : EarlyExitingBuilder.createNot(CondOfEarlyExitingVPBB);
+
assert((isa<VPIRValue>(CondOfEarlyExitingVPBB) ||
!VPDT.properlyDominates(EarlyExitingVPBB, LatchVPBB) ||
VPDT.properlyDominates(
@@ -3416,6 +3471,11 @@ bool VPlanTransforms::handleUncountableEarlyExits(
DispatchBuilder.setInsertPoint(CurrentBB);
}
+ VPDT.recalculate(Plan);
+ SmallVector<VPBasicBlock *> LiveOutVPBBs = {MiddleVPBB};
+ append_range(LiveOutVPBBs, VectorEarlyExitVPBBs);
+ repairEarlyExitSSA(Plan, VPDT, Exits, LatchVPBB, LiveOutVPBBs);
+
return true;
}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index c01ddfbd24263..58e23806bbcc1 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -400,7 +400,6 @@ static bool preservesUniformity(unsigned Opcode) {
case Instruction::Select:
case VPInstruction::Not:
case VPInstruction::Broadcast:
- case VPInstruction::MaskedCond:
case VPInstruction::PtrAdd:
return true;
default:
diff --git a/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp b/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
index a4d6cc0ecb480..c0595ac8704c3 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
@@ -251,11 +251,6 @@ bool VPlanVerifier::verifyVPBasicBlock(const VPBasicBlock *VPBB) {
return false;
}
- // MaskedCond may be used from blocks it don't dominate; the block will be
- // linearized and it will dominate its users after linearization.
- if (match(&R, m_VPInstruction<VPInstruction::MaskedCond>()))
- continue;
-
for (const VPUser *U : V->users()) {
auto *UI = cast<VPRecipeBase>(U);
if (isa<VPIRPhi>(UI) &&
@@ -300,19 +295,6 @@ bool VPlanVerifier::verifyVPBasicBlock(const VPBasicBlock *VPBB) {
continue;
}
- // Recipes in blocks with a MaskedCond may be used in exit blocks; the
- // block will be linearized and its recipes will dominate their users
- // after linearization.
- bool BlockHasMaskedCond = any_of(*VPBB, [](const VPRecipeBase &R) {
- return match(&R, m_VPInstruction<VPInstruction::MaskedCond>());
- });
- if (BlockHasMaskedCond &&
- any_of(VPBB->getPlan()->getExitBlocks(), [UI](VPIRBasicBlock *EB) {
- return is_contained(EB->getPredecessors(), UI->getParent());
- })) {
- continue;
- }
-
errs() << "Use before def!\n";
#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
VPSlotTracker Tracker(VPBB->getPlan());
diff --git a/llvm/test/Transforms/LoopVectorize/predicated-early-exits-interleave.ll b/llvm/test/Transforms/LoopVectorize/predicated-early-exits-interleave.ll
index df24b2838f6c4..d07ea380f55b8 100644
--- a/llvm/test/Transforms/LoopVectorize/predicated-early-exits-interleave.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicated-early-exits-interleave.ll
@@ -185,10 +185,8 @@ define i64 @three_early_exits() {
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[GEP_A]], i64 4
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[GEP_A]], align 1
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], splat (i8 -42)
-; CHECK-NEXT: [[TMP3:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD1]], splat (i8 -42)
-; CHECK-NEXT: [[TMP4:%.*]] = xor <4 x i1> [[TMP2]], splat (i1 true)
-; CHECK-NEXT: [[TMP5:%.*]] = xor <4 x i1> [[TMP3]], splat (i1 true)
+; CHECK-NEXT: [[TMP4:%.*]] = icmp sge <4 x i8> [[WIDE_LOAD]], splat (i8 -42)
+; CHECK-NEXT: [[TMP5:%.*]] = icmp sge <4 x i8> [[WIDE_LOAD1]], splat (i8 -42)
; CHECK-NEXT: [[TMP6:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], splat (i8 42)
; CHECK-NEXT: [[TMP7:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD1]], splat (i8 42)
; CHECK-NEXT: [[TMP8:%.*]] = xor <4 x i1> [[TMP6]], splat (i1 true)
@@ -201,8 +199,6 @@ define i64 @three_early_exits() {
; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP13]], align 1
; CHECK-NEXT: [[TMP14:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
; CHECK-NEXT: [[TMP15:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD1]], [[WIDE_LOAD3]]
-; CHECK-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[TMP17:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP15]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP18:%.*]] = select <4 x i1> [[TMP4]], <4 x i1> [[TMP6]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP19:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP20:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
@@ -211,16 +207,18 @@ define i64 @three_early_exits() {
; CHECK-NEXT: [[WIDE_LOAD5:%.*]] = load <4 x i8>, ptr [[TMP21]], align 1
; CHECK-NEXT: [[TMP22:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD4]]
; CHECK-NEXT: [[TMP23:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD1]], [[WIDE_LOAD5]]
-; CHECK-NEXT: [[TMP24:%.*]] = select <4 x i1> [[TMP18]], <4 x i1> [[TMP22]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP19]], <4 x i1> [[TMP23]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP26:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[TMP27:%.*]] = getelementptr inbounds i8, ptr [[TMP26]], i64 4
; CHECK-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i8>, ptr [[TMP26]], align 1
; CHECK-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i8>, ptr [[TMP27]], align 1
; CHECK-NEXT: [[TMP28:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD6]]
; CHECK-NEXT: [[TMP29:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD1]], [[WIDE_LOAD7]]
-; CHECK-NEXT: [[TMP30:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP28]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[TMP31:%.*]] = select <4 x i1> [[TMP3]], <4 x i1> [[TMP29]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP17:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP15]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP24:%.*]] = select <4 x i1> [[TMP18]], <4 x i1> [[TMP22]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP19]], <4 x i1> [[TMP23]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP30:%.*]] = select <4 x i1> [[TMP4]], <4 x i1> zeroinitializer, <4 x i1> [[TMP28]]
+; CHECK-NEXT: [[TMP31:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> zeroinitializer, <4 x i1> [[TMP29]]
; CHECK-NEXT: [[TMP32:%.*]] = select <4 x i1> [[TMP16]], <4 x i1> splat (i1 true), <4 x i1> [[TMP24]]
; CHECK-NEXT: [[TMP33:%.*]] = select <4 x i1> [[TMP17]], <4 x i1> splat (i1 true), <4 x i1> [[TMP25]]
; CHECK-NEXT: [[TMP34:%.*]] = select <4 x i1> [[TMP32]], <4 x i1> splat (i1 true), <4 x i1> [[TMP30]]
diff --git a/llvm/test/Transforms/LoopVectorize/predicated-multiple-exits.ll b/llvm/test/Transforms/LoopVectorize/predicated-multiple-exits.ll
index f672f649915a4..76b693331be28 100644
--- a/llvm/test/Transforms/LoopVectorize/predicated-multiple-exits.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicated-multiple-exits.ll
@@ -17,15 +17,14 @@ define i64 @diamond_with_2_early_exits() {
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr @A, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = xor <4 x i1> [[TMP1]], splat (i1 true)
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP4]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[GEP_B]], align 1
; CHECK-NEXT: [[TMP7:%.*]] = zext <4 x i8> [[WIDE_LOAD2]] to <4 x i64>
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> zeroinitializer, <4 x i1> [[TMP4]]
; CHECK-NEXT: [[TMP9:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP8]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP10:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> splat (i1 true), <4 x i1> [[TMP9]]
; CHECK-NEXT: [[TMP11:%.*]] = freeze <4 x i1> [[TMP10]]
@@ -94,24 +93,23 @@ define i64 @three_early_exits() {
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[LOOP_HEADER]] ], [ [[INDEX_NEXT:%.*]], %[[CHECK_B:.*]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i8, ptr @A, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[GEP_A]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], splat (i8 -42)
-; CHECK-NEXT: [[TMP2:%.*]] = xor <4 x i1> [[TMP1]], splat (i1 true)
+; CHECK-NEXT: [[TMP2:%.*]] = icmp sge <4 x i8> [[WIDE_LOAD]], splat (i8 -42)
; CHECK-NEXT: [[TMP3:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], splat (i8 42)
; CHECK-NEXT: [[TMP4:%.*]] = xor <4 x i1> [[TMP3]], splat (i1 true)
; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP4]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i8, ptr @D, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[TMP6]], align 1
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP8:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP9:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP3]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP10]], align 1
; CHECK-NEXT: [[TMP11:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
-; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP11]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP13:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP13]], align 1
; CHECK-NEXT: [[TMP14:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD3]]
-; CHECK-NEXT: [[TMP15:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP8:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP11]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP15:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> zeroinitializer, <4 x i1> [[TMP14]]
; CHECK-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP8]], <4 x i1> splat (i1 true), <4 x i1> [[TMP12]]
; CHECK-NEXT: [[TMP17:%.*]] = select <4 x i1> [[TMP16]], <4 x i1> splat (i1 true), <4 x i1> [[TMP15]]
; CHECK-NEXT: [[TMP18:%.*]] = freeze <4 x i1> [[TMP17]]
@@ -193,11 +191,9 @@ define i64 @nested_diamond_inner_exits() {
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr @A, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = xor <4 x i1> [[TMP1]], splat (i1 true)
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr @D, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP4]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[GEP_B]], align 1
; CHECK-NEXT: [[TMP7:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD2]], zeroinitializer
@@ -206,9 +202,10 @@ define i64 @nested_diamond_inner_exits() {
; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP10]], align 1
; CHECK-NEXT: [[TMP11:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD3]]
-; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP11]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP13:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP14:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> zeroinitializer, <4 x i1> [[TMP4]]
+; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP11]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP15:%.*]] = select <4 x i1> [[TMP13]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> splat (i1 true), <4 x i1> [[TMP12]]
; CHECK-NEXT: [[TMP17:%.*]] = select <4 x i1> [[TMP16]], <4 x i1> splat (i1 true), <4 x i1> [[TMP15]]
@@ -297,14 +294,14 @@ define i64 @chain_of_3_exits() {
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[GEP_B]], align 1
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP4:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP3]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[GEP_C:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[GEP_C]], align 1
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
-; CHECK-NEXT: [[TMP7:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP6]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds i8, ptr @D, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP8]], align 1
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD3]]
+; CHECK-NEXT: [[TMP4:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP3]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP7:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP6]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP10:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP9]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP11:%.*]] = select <4 x i1> [[TMP4]], <4 x i1> splat (i1 true), <4 x i1> [[TMP7]]
; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> splat (i1 true), <4 x i1> [[TMP10]]
@@ -383,26 +380,24 @@ define i64 @four_exits_2x2_diamond() {
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr @A, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = xor <4 x i1> [[TMP1]], splat (i1 true)
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP4]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[GEP_B]], align 1
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> zeroinitializer, <4 x i1> [[TMP4]]
; CHECK-NEXT: [[TMP8:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i8, ptr @D, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP9]], align 1
; CHECK-NEXT: [[TMP10:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD3]], zeroinitializer
-; CHECK-NEXT: [[TMP11:%.*]] = xor <4 x i1> [[TMP10]], splat (i1 true)
; CHECK-NEXT: [[TMP12:%.*]] = icmp ne <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD3]]
-; CHECK-NEXT: [[TMP13:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP12]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP14:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD3]]
-; CHECK-NEXT: [[TMP15:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[PREDPHI5:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> zeroinitializer, <4 x i1> [[TMP12]]
+; CHECK-NEXT: [[PREDPHI6:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> splat (i1 true), <4 x i1> [[TMP8]]
-; CHECK-NEXT: [[TMP17:%.*]] = select <4 x i1> [[TMP16]], <4 x i1> splat (i1 true), <4 x i1> [[TMP13]]
-; CHECK-NEXT: [[TMP18:%.*]] = select <4 x i1> [[TMP17]], <4 x i1> splat (i1 true), <4 x i1> [[TMP15]]
+; CHECK-NEXT: [[TMP11:%.*]] = select <4 x i1> [[TMP16]], <4 x i1> splat (i1 true), <4 x i1> [[PREDPHI5]]
+; CHECK-NEXT: [[TMP18:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> splat (i1 true), <4 x i1> [[PREDPHI6]]
; CHECK-NEXT: [[TMP19:%.*]] = freeze <4 x i1> [[TMP18]]
; CHECK-NEXT: [[CMP1A:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP19]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
@@ -420,7 +415,7 @@ define i64 @four_exits_2x2_diamond() {
; CHECK-NEXT: [[TMP23:%.*]] = extractelement <4 x i1> [[TMP8]], i64 [[FIRST_ACTIVE_LANE]]
; CHECK-NEXT: br i1 [[TMP23]], label %[[VECTOR_EARLY_EXIT_1:.*]], label %[[LOOP_LATCH:.*]]
; CHECK: [[LOOP_LATCH]]:
-; CHECK-NEXT: [[TMP24:%.*]] = extractelement <4 x i1> [[TMP13]], i64 [[FIRST_ACTIVE_LANE]]
+; CHECK-NEXT: [[TMP24:%.*]] = extractelement <4 x i1> [[PREDPHI5]], i64 [[FIRST_ACTIVE_LANE]]
; CHECK-NEXT: br i1 [[TMP24]], label %[[VECTOR_EARLY_EXIT_2:.*]], label %[[VECTOR_EARLY_EXIT_3:.*]]
; CHECK: [[VECTOR_EARLY_EXIT_3]]:
; CHECK-NEXT: br label %[[LOOP_END1]]
@@ -577,10 +572,9 @@ define i64 @diamond_exits_overlapping_conditions() {
; CHECK-NEXT: [[GEP_C:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[GEP_C]], align 1
; CHECK-NEXT: [[TMP3:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT: [[TMP4:%.*]] = xor <4 x i1> [[TMP3]], splat (i1 true)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
-; CHECK-NEXT: [[TMP6:%.*]] = select <4 x i1> [[TMP4]], <4 x i1> [[TMP5]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP6:%.*]] = select <4 x i1> [[TMP3]], <4 x i1> zeroinitializer, <4 x i1> [[TMP5]]
; CHECK-NEXT: [[TMP8:%.*]] = select <4 x i1> [[TMP3]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP9:%.*]] = select <4 x i1> [[TMP6]], <4 x i1> splat (i1 true), <4 x i1> [[TMP8]]
; CHECK-NEXT: [[TMP10:%.*]] = freeze <4 x i1> [[TMP9]]
diff --git a/llvm/test/Transforms/LoopVectorize/single_early_exit.ll b/llvm/test/Transforms/LoopVectorize/single_early_exit.ll
index f2f563a9032f8..459cb9d7be56a 100644
--- a/llvm/test/Transforms/LoopVectorize/single_early_exit.ll
+++ b/llvm/test/Transforms/LoopVectorize/single_early_exit.ll
@@ -620,6 +620,67 @@ exit:
%res = phi i64 [ -1, %entry ], [ -2, %then ], [ 0, %loop.latch ], [ %iv, %loop.header ]
ret i64 %res
}
+
+define i64 @same_exit_block_phi_of_consts_iv_next_and() {
+; CHECK-LABEL: define i64 @same_exit_block_phi_of_consts_iv_next_and() {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[P1:%.*]] = alloca [1024 x i8], align 1
+; CHECK-NEXT: [[P2:%.*]] = alloca [1024 x i8], align 1
+; CHECK-NEXT: call void @init_mem(ptr [[P1]], i64 1024)
+; CHECK-NEXT: call void @init_mem(ptr [[P2]], i64 1024)
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX1:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT3:%.*]], [[VECTOR_BODY_INTERIM:%.*]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = add i64 3, [[INDEX1]]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[P1]], i64 [[TMP0]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[P2]], i64 [[TMP0]]
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
+; CHECK-NEXT: [[TMP3:%.*]] = icmp ne <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP4:%.*]] = freeze <4 x i1> [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP4]])
+; CHECK-NEXT: [[INDEX_NEXT3]] = add nuw i64 [[INDEX1]], 4
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT3]], 64
+; CHECK-NEXT: br i1 [[TMP5]], label [[VECTOR_EARLY_EXIT:%.*]], label [[VECTOR_BODY_INTERIM]]
+; CHECK: vector.body.interim:
+; CHECK-NEXT: br i1 [[TMP6]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: br label [[LOOP_END:%.*]]
+; CHECK: vector.early.exit:
+; CHECK-NEXT: br label [[LOOP_END]]
+; CHECK: loop.end:
+; CHECK-NEXT: [[RETVAL:%.*]] = phi i64 [ 0, [[VECTOR_EARLY_EXIT]] ], [ 1, [[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i64 [[RETVAL]]
+;
+entry:
+ %p1 = alloca [1024 x i8]
+ %p2 = alloca [1024 x i8]
+ call void @init_mem(ptr %p1, i64 1024)
+ call void @init_mem(ptr %p2, i64 1024)
+ br label %loop
+
+loop:
+ %index = phi i64 [ %index.next, %loop.inc ], [ 3, %entry ]
+ %arrayidx = getelementptr inbounds i8, ptr %p1, i64 %index
+ %ld1 = load i8, ptr %arrayidx, align 1
+ %arrayidx1 = getelementptr inbounds i8, ptr %p2, i64 %index
+ %ld2 = load i8, ptr %arrayidx1, align 1
+ %cmp3 = icmp eq i8 %ld1, %ld2
+ br i1 %cmp3, label %loop.inc, label %loop.end
+
+loop.inc:
+ %index.next = add i64 %index, 1
+ %index.next.and = and i64 %index.next, 4294967295
+ %exitcond = icmp ne i64 %index.next.and, 67
+ br i1 %exitcond, label %loop, label %loop.end
+
+loop.end:
+ %retval = phi i64 [ 0, %loop ], [ 1, %loop.inc ]
+ ret i64 %retval
+}
+
;.
; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
@@ -634,4 +695,5 @@ exit:
; CHECK: [[LOOP10]] = distinct !{[[LOOP10]], [[META2]], [[META1]]}
; CHECK: [[LOOP11]] = distinct !{[[LOOP11]], [[META1]], [[META2]]}
; CHECK: [[LOOP12]] = distinct !{[[LOOP12]], [[META2]], [[META1]]}
+; CHECK: [[LOOP13]] = distinct !{[[LOOP13]], [[META1]], [[META2]]}
;.
>From 0225d332e544a8ecc084e4dc36172699efe18108 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Tue, 9 Jun 2026 18:10:38 +0800
Subject: [PATCH 7/9] Address review comments
* Explain when to use repairSSA
* repairSSA -> repairSSAImpl
---
.../lib/Transforms/Vectorize/VPlanTransforms.cpp | 16 +++++++++-------
1 file changed, 9 insertions(+), 7 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index c4b55e029a598..c7a10996f4420 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -2997,9 +2997,10 @@ struct EarlyExitInfo {
VPIRBasicBlock *EarlyExitVPBB;
VPValue *CondToExit;
};
-static VPValue *repairSSA(VPValue *Src, VPBasicBlock *SrcVPBB, VPValue *Other,
- VPBasicBlock *VPBB, VPDominatorTree &VPDT,
- DenseMap<VPBlockBase *, VPPhi *> &Phis) {
+static VPValue *repairSSAImpl(VPValue *Src, VPBasicBlock *SrcVPBB,
+ VPValue *Other, VPBasicBlock *VPBB,
+ VPDominatorTree &VPDT,
+ DenseMap<VPBlockBase *, VPPhi *> &Phis) {
if (VPDT.dominates(SrcVPBB, VPBB))
return Src;
@@ -3010,8 +3011,8 @@ static VPValue *repairSSA(VPValue *Src, VPBasicBlock *SrcVPBB, VPValue *Other,
SmallVector<VPValue *> InVals;
for (auto *Pred : VPBB->predecessors())
- InVals.push_back(
- repairSSA(Src, SrcVPBB, Other, cast<VPBasicBlock>(Pred), VPDT, Phis));
+ InVals.push_back(repairSSAImpl(Src, SrcVPBB, Other,
+ cast<VPBasicBlock>(Pred), VPDT, Phis));
if (all_equal(InVals))
return InVals[0];
@@ -3022,11 +3023,12 @@ static VPValue *repairSSA(VPValue *Src, VPBasicBlock *SrcVPBB, VPValue *Other,
/// Insert phi nodes to maintain SSA starting from \p VPBB, such that the
/// resulting value is \p \Src on all paths that go through \p SrcVPBB, and \p
-/// Other otherwise.
+/// Other otherwise. Use if the CFG has been modified such that a def no longer
+/// dominates all its uses.
static VPValue *repairSSA(VPValue *Src, VPBasicBlock *SrcVPBB, VPValue *Other,
VPBasicBlock *VPBB, VPDominatorTree &VPDT) {
DenseMap<VPBlockBase *, VPPhi *> Phis;
- return repairSSA(Src, SrcVPBB, Other, VPBB, VPDT, Phis);
+ return repairSSAImpl(Src, SrcVPBB, Other, VPBB, VPDT, Phis);
}
// After handling early exits, the CondToExits and live outs may no longer be in
>From 33747bb7d973d4de39dcd758e9086294c0e9c768 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Thu, 23 Jul 2026 17:27:35 +0800
Subject: [PATCH 8/9] Rework reconstruction to align with Braun et al algorithm
---
.../Transforms/Vectorize/VPlanTransforms.cpp | 73 ++++++-------------
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 35 +++++++++
llvm/lib/Transforms/Vectorize/VPlanUtils.h | 5 ++
3 files changed, 61 insertions(+), 52 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index c7a10996f4420..5a21b2d89b3ea 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -2997,60 +2997,29 @@ struct EarlyExitInfo {
VPIRBasicBlock *EarlyExitVPBB;
VPValue *CondToExit;
};
-static VPValue *repairSSAImpl(VPValue *Src, VPBasicBlock *SrcVPBB,
- VPValue *Other, VPBasicBlock *VPBB,
- VPDominatorTree &VPDT,
- DenseMap<VPBlockBase *, VPPhi *> &Phis) {
-
- if (VPDT.dominates(SrcVPBB, VPBB))
- return Src;
- if (VPDT.dominates(VPBB, SrcVPBB))
- return Other;
- if (VPPhi *Phi = Phis.lookup(VPBB))
- return Phi;
-
- SmallVector<VPValue *> InVals;
- for (auto *Pred : VPBB->predecessors())
- InVals.push_back(repairSSAImpl(Src, SrcVPBB, Other,
- cast<VPBasicBlock>(Pred), VPDT, Phis));
- if (all_equal(InVals))
- return InVals[0];
-
- VPPhi *Phi = VPBuilder(VPBB, VPBB->getFirstNonPhi()).createScalarPhi(InVals);
- Phis[VPBB] = Phi;
- return Phi;
-}
-
-/// Insert phi nodes to maintain SSA starting from \p VPBB, such that the
-/// resulting value is \p \Src on all paths that go through \p SrcVPBB, and \p
-/// Other otherwise. Use if the CFG has been modified such that a def no longer
-/// dominates all its uses.
-static VPValue *repairSSA(VPValue *Src, VPBasicBlock *SrcVPBB, VPValue *Other,
- VPBasicBlock *VPBB, VPDominatorTree &VPDT) {
- DenseMap<VPBlockBase *, VPPhi *> Phis;
- return repairSSAImpl(Src, SrcVPBB, Other, VPBB, VPDT, Phis);
-}
// After handling early exits, the CondToExits and live outs may no longer be in
-// SSA if their defining blocks are predicated, so insert phis to repair them.
-static void repairEarlyExitSSA(VPlan &Plan, VPDominatorTree &VPDT,
- ArrayRef<EarlyExitInfo> Exits,
- VPBasicBlock *LatchVPBB,
- ArrayRef<VPBasicBlock *> LiveOutVPBBs) {
- // Repair all CondToExits. The condition is false on any path that doesn't go
- // through the exiting block.
+// SSA if their defining blocks are predicated. Reconstruct by inserting phis.
+static void reconstructEarlyExitSSA(VPlan &Plan, VPDominatorTree &VPDT,
+ ArrayRef<EarlyExitInfo> Exits,
+ VPBasicBlock *HeaderVPBB,
+ VPBasicBlock *LatchVPBB,
+ ArrayRef<VPBasicBlock *> LiveOutVPBBs) {
+ // Reconstruct all CondToExits. The condition is false on any path that
+ // doesn't go through the exiting block.
for (auto [EarlyExitingVPBB, _, CondToExit] : Exits) {
- VPValue *Repaired = repairSSA(CondToExit, EarlyExitingVPBB, Plan.getFalse(),
- LatchVPBB, VPDT);
+ VPValue *New = vputils::reconstructSSA(
+ {{EarlyExitingVPBB, CondToExit}, {HeaderVPBB, Plan.getFalse()}},
+ LatchVPBB);
- CondToExit->replaceUsesWithIf(Repaired, [&](VPUser &U, unsigned I) {
+ CondToExit->replaceUsesWithIf(New, [&](VPUser &U, unsigned I) {
auto &R = cast<VPRecipeBase>(U);
return VPDT.dominates(LatchVPBB, R.getParent()) &&
- R.getVPSingleValue() != Repaired;
+ R.getVPSingleValue() != New;
});
}
- // Repair any live outs. The value is poison on any path that didn't pass
+ // Reconstruct any live outs. The value is poison on any path that didn't pass
// through the def's block.
for (VPBasicBlock *LiveOutVPBB : LiveOutVPBBs)
for (VPRecipeBase &R : *LiveOutVPBB) {
@@ -3059,12 +3028,11 @@ static void repairEarlyExitSSA(VPlan &Plan, VPDominatorTree &VPDT,
m_CombineOr(m_ExtractLastPart(m_VPValue(LiveOut)),
m_ExtractLane(m_VPValue(), m_VPValue(LiveOut)))))
continue;
- VPValue *Poison =
- Plan.getOrAddLiveIn(PoisonValue::get(LiveOut->getScalarType()));
- VPValue *Repaired =
- repairSSA(LiveOut, LiveOut->getDefiningRecipe()->getParent(), Poison,
- LatchVPBB, VPDT);
- R.replaceUsesOfWith(LiveOut, Repaired);
+ VPValue *New = vputils::reconstructSSA(
+ {{LiveOut->getDefiningRecipe()->getParent(), LiveOut},
+ {HeaderVPBB, Plan.getPoison(LiveOut->getScalarType())}},
+ LatchVPBB);
+ R.replaceUsesOfWith(LiveOut, New);
}
}
@@ -3476,7 +3444,8 @@ bool VPlanTransforms::handleUncountableEarlyExits(
VPDT.recalculate(Plan);
SmallVector<VPBasicBlock *> LiveOutVPBBs = {MiddleVPBB};
append_range(LiveOutVPBBs, VectorEarlyExitVPBBs);
- repairEarlyExitSSA(Plan, VPDT, Exits, LatchVPBB, LiveOutVPBBs);
+ reconstructEarlyExitSSA(Plan, VPDT, Exits, HeaderVPBB, LatchVPBB,
+ LiveOutVPBBs);
return true;
}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index 58e23806bbcc1..4445282246360 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -1130,3 +1130,38 @@ void vputils::detail::pullOutPermutationsImpl(
}
}
}
+
+/// Implements the algorithm described in "Simple and Efficient Construction of
+/// Static Single Assignment Form" by Braun et al.
+static VPValue *reconstructSSAImpl(VPBasicBlock *VPBB,
+ DenseMap<VPBlockBase *, VPValue *> &Defs) {
+ if (VPValue *Def = Defs.lookup(VPBB))
+ return Def;
+
+ if (VPBlockBase *Pred = VPBB->getSinglePredecessor())
+ return reconstructSSAImpl(cast<VPBasicBlock>(Pred), Defs);
+
+ // Multiple predecessors, create a join.
+ Type *Ty = Defs.begin()->second->getScalarType();
+ auto *Phi = new VPPhi({}, {}, {}, "", Ty);
+ VPBB->insert(Phi, VPBB->getFirstNonPhi());
+ Defs[VPBB] = Phi;
+ for (auto *Pred : VPBB->predecessors())
+ Phi->addIncoming(reconstructSSAImpl(cast<VPBasicBlock>(Pred), Defs));
+
+ // Fold away trivial phis.
+ if (all_equal(Phi->incoming_values())) {
+ VPValue *Common = Phi->getIncomingValue(0);
+ Phi->replaceAllUsesWith(Common);
+ Phi->eraseFromParent();
+ Defs[VPBB] = Common;
+ return Common;
+ }
+
+ return Phi;
+}
+
+VPValue *vputils::reconstructSSA(DenseMap<VPBlockBase *, VPValue *> Defs,
+ VPBasicBlock *VPBB) {
+ return reconstructSSAImpl(VPBB, Defs);
+}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.h b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
index 643181bd7fafb..42604ca38463b 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.h
@@ -215,6 +215,11 @@ SmallVector<VPUser *> collectUsersRecursively(VPValue *V);
VPIRValue *tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef<VPValue *> Operands,
const DataLayout &DL);
+/// Insert phis to reconstruct SSA starting from \p VPBB, looking up
+/// definitions from \p Defs. Use if the CFG has been modified such that a def
+/// no longer dominates all its uses.
+VPValue *reconstructSSA(DenseMap<VPBlockBase *, VPValue *> Defs,
+ VPBasicBlock *VPBB);
namespace detail {
/// Template-independent implementation for pullOutPermutations.
>From d889e5cc6a8711b1ba2837297a50ee55d4fc57ca Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Wed, 10 Jun 2026 15:41:56 +0800
Subject: [PATCH 9/9] [VPlan] Use CFG to mask early exit loops with
side-effects. NFC
---
.../Transforms/Vectorize/VPlanTransforms.cpp | 39 ++++++++++++++++---
1 file changed, 33 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 5a21b2d89b3ea..9df4cbe9e5848 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -3036,6 +3036,26 @@ static void reconstructEarlyExitSSA(VPlan &Plan, VPDominatorTree &VPDT,
}
}
+/// Splits \p LatchVPBB so it only contains the IV increment recipes.
+static VPBasicBlock *splitLatchAtIVInc(VPBasicBlock *LatchVPBB) {
+ auto It = std::prev(LatchVPBB->getTerminator()->getIterator());
+ while (!match(
+ It->getVPSingleValue(),
+ m_CombineOr(m_Add(m_Isa<VPWidenIntOrFpInductionRecipe>(), m_LiveIn()),
+ m_Sub(m_Isa<VPWidenIntOrFpInductionRecipe>(), m_LiveIn()),
+ m_VPInstruction<Instruction::GetElementPtr>(
+ m_Isa<VPWidenPointerInductionRecipe>(), m_LiveIn())))) {
+ assert(!It->mayReadOrWriteMemory() && !It->mayHaveSideEffects() &&
+ "Instruction with side effects between IV increment and branch?");
+ if (It == LatchVPBB->begin())
+ return LatchVPBB;
+ It = std::prev(It);
+ }
+ LatchVPBB = LatchVPBB->splitAt(It);
+ LatchVPBB->setName("vector.latch");
+ return LatchVPBB;
+}
+
/// Update \p Plan to mask memory operations in the loop based on whether the
/// early exit is taken or not.
///
@@ -3079,6 +3099,8 @@ static bool handleUncountableExitsWithSideEffects(
VPBasicBlock *HeaderVPBB, VPBasicBlock *LatchVPBB, VPBasicBlock *MiddleVPBB,
Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT,
AssumptionCache *AC) {
+ VPBasicBlock *OrigLatchVPBB = LatchVPBB;
+ LatchVPBB = splitLatchAtIVInc(LatchVPBB);
// Disconnect early exiting blocks from successors, remove branches. We
// currently don't support multiple uses for recipes involved in creating
@@ -3177,16 +3199,18 @@ static bool handleUncountableExitsWithSideEffects(
VPValue *Mask = MaskBuilder.createNaryOp(VPInstruction::ActiveLaneMask,
{Zero, FirstActive, ALMMultiplier},
DebugLoc(), "uncountable.exit.mask");
+ VPInstruction *Branch =
+ MaskBuilder.createNaryOp(VPInstruction::BranchOnCond, Mask);
+ HeaderVPBB->splitAt(std::next(Branch->getIterator()));
+ VPBlockUtils::connectBlocks(HeaderVPBB, LatchVPBB);
+ VPDT.recalculate(Plan);
- // Convert all other memory operations to use the mask.
+ // TODO: Remove restriction on conditional memory operations in the loop.
for (VPBasicBlock *VPBB : vp_rpo_plain_cfg_loop_body(HeaderVPBB))
for (VPRecipeBase &R : *VPBB)
- if (R.mayReadOrWriteMemory() && &R != Load) {
- // TODO: Handle conditional memory operations in the loop.
- if (!VPDT.dominates(R.getParent(), LatchVPBB))
+ if (R.mayReadOrWriteMemory() && &R != Load)
+ if (!VPDT.dominates(R.getParent(), OrigLatchVPBB))
return false;
- cast<VPInstruction>(&R)->addMask(Mask);
- }
// Update middle block branch to compare (IV + however many lanes were active)
// against the full trip count, since we may be exiting the vector loop early.
@@ -3216,6 +3240,9 @@ static bool handleUncountableExitsWithSideEffects(
m_VPInstruction<VPInstruction::ExitingIVValue>(m_Specific(IV))) &&
"Continuing from different IV");
ContinueIV->setOperand(0, ExitIV);
+
+ reconstructEarlyExitSSA(Plan, VPDT, Exits, HeaderVPBB, LatchVPBB, MiddleVPBB);
+
return true;
}
More information about the llvm-commits
mailing list