[llvm] [VPlan] Model control flow of early exits and preserve SSA (PR #201784)
Luke Lau via llvm-commits
llvm-commits at lists.llvm.org
Mon Jun 8 00:33:21 PDT 2026
https://github.com/lukel97 updated https://github.com/llvm/llvm-project/pull/201784
>From f16c94b767636408219f3b54a9d45c50774d79b7 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Mon, 8 Jun 2026 14:57:32 +0800
Subject: [PATCH 1/8] Precommit test
---
.../LoopVectorize/VPlan/predicator.ll | 108 ++++++++++++++++++
1 file changed, 108 insertions(+)
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index 2e955de703cdf..e8000cc8a25c3 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -638,3 +638,111 @@ bb3:
exit:
ret void
}
+
+define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
+; CHECK-LABEL: VPlan for loop in 'blend_chain_non_trivial'
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION nuw nsw ir<0>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK-NEXT: EMIT-SCALAR ir<%lb> = load ir<%b>
+; CHECK-NEXT: EMIT ir<%v1> = add ir<%iv>, ir<%lb>
+; CHECK-NEXT: EMIT ir<%v2> = mul ir<%iv>, ir<3>
+; CHECK-NEXT: EMIT ir<%gep> = getelementptr ir<%a>, ir<%iv>
+; CHECK-NEXT: EMIT ir<%c0> = icmp sle ir<%iv>, ir<0>
+; CHECK-NEXT: Successor(s): if.a
+; CHECK-EMPTY:
+; CHECK-NEXT: if.a:
+; CHECK-NEXT: EMIT ir<%ca> = icmp sle ir<%iv>, ir<8>, ir<%c0>
+; CHECK-NEXT: Successor(s): if.a.inner
+; CHECK-EMPTY:
+; CHECK-NEXT: if.a.inner:
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = logical-and ir<%c0>, ir<%ca>
+; CHECK-NEXT: Successor(s): merge.a.inner
+; CHECK-EMPTY:
+; CHECK-NEXT: merge.a.inner:
+; CHECK-NEXT: Successor(s): merge.a
+; CHECK-EMPTY:
+; CHECK-NEXT: merge.a:
+; CHECK-NEXT: EMIT ir<%d0> = icmp sgt ir<%iv>, ir<0>
+; CHECK-NEXT: Successor(s): if.b
+; CHECK-EMPTY:
+; CHECK-NEXT: if.b:
+; CHECK-NEXT: EMIT ir<%cb> = icmp sle ir<%iv>, ir<16>, ir<%d0>
+; CHECK-NEXT: Successor(s): if.b.inner
+; CHECK-EMPTY:
+; CHECK-NEXT: if.b.inner:
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
+; CHECK-NEXT: Successor(s): merge.b.inner
+; CHECK-EMPTY:
+; CHECK-NEXT: merge.b.inner:
+; CHECK-NEXT: Successor(s): merge.b
+; CHECK-EMPTY:
+; CHECK-NEXT: merge.b:
+; CHECK-NEXT: EMIT ir<%sum> = add ir<%v1>, ir<%v2>
+; CHECK-NEXT: EMIT store ir<%sum>, ir<%gep>
+; CHECK-NEXT: Successor(s): loop.latch
+; CHECK-EMPTY:
+; CHECK-NEXT: loop.latch:
+; CHECK-NEXT: EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
+; CHECK-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<128>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK-NEXT: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block
+;
+entry:
+ br label %loop.header
+
+loop.header:
+ %iv = phi i64 [ 0, %entry ], [ %iv.next, %loop.latch ]
+ %lb = load i64, ptr %b
+ %v1 = add i64 %iv, %lb
+ %v2 = mul i64 %iv, 3
+ %gep = getelementptr i64, ptr %a, i64 %iv
+ %c0 = icmp sle i64 %iv, 0
+ br i1 %c0, label %if.a, label %merge.a
+
+if.a:
+ %ca = icmp sle i64 %iv, 8
+ br i1 %ca, label %if.a.inner, label %merge.a.inner
+
+if.a.inner:
+ br label %merge.a.inner
+
+merge.a.inner:
+ %blend.a.inner = phi i64 [ %v1, %if.a ], [ %v1, %if.a.inner ]
+ br label %merge.a
+
+merge.a:
+ %blend.a = phi i64 [ %v1, %loop.header ], [ %blend.a.inner, %merge.a.inner ]
+ %d0 = icmp sgt i64 %iv, 0
+ br i1 %d0, label %if.b, label %merge.b
+
+if.b:
+ %cb = icmp sle i64 %iv, 16
+ br i1 %cb, label %if.b.inner, label %merge.b.inner
+
+if.b.inner:
+ br label %merge.b.inner
+
+merge.b.inner:
+ %blend.b.inner = phi i64 [ %v2, %if.b ], [ %v2, %if.b.inner ]
+ br label %merge.b
+
+merge.b:
+ %blend.b = phi i64 [ %v2, %merge.a ], [ %blend.b.inner, %merge.b.inner ]
+ %sum = add i64 %blend.a, %blend.b
+ store i64 %sum, ptr %gep
+ br label %loop.latch
+
+loop.latch:
+ %iv.next = add nuw nsw i64 %iv, 1
+ %ec = icmp eq i64 %iv.next, 128
+ br i1 %ec, label %exit, label %loop.header
+
+exit:
+ ret void
+}
>From fd134c39c732cd16fa7f30ac37b06703aafbc0a3 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Tue, 2 Jun 2026 18:15:25 +0800
Subject: [PATCH 2/8] Insert blends in post order
---
llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp | 15 ++++++++++++---
.../Transforms/LoopVectorize/VPlan/predicator.ll | 8 ++++++--
2 files changed, 18 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 2717b80e2eeaa..b01fef556f8ed 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -69,6 +69,8 @@ class VPPredicator {
return EdgeMaskCache[{Src, Dst}] = Mask;
}
+ DenseMap<const VPBasicBlock *, VPBasicBlock::iterator> InsertPoints;
+
public:
VPPredicator(VPlan &Plan) : VPDT(Plan), VPPDT(Plan) {}
@@ -136,6 +138,10 @@ void VPPredicator::createBlockInMask(VPBasicBlock *VPBB) {
// Start inserting after the block's phis, which be replaced by blends later.
Builder.setInsertPoint(VPBB, VPBB->getFirstNonPhi());
+ // Keep track of where in VPBB we are inserting the masks into.
+ scope_exit UpdateInsertPoint(
+ [this, &VPBB]() { InsertPoints[VPBB] = Builder.getInsertPoint(); });
+
// Reuse the mask of the immediate dominator if the VPBB post-dominates the
// immediate dominator.
auto *IDom = VPDT.getNode(VPBB)->getIDom();
@@ -225,6 +231,7 @@ void VPPredicator::createSwitchEdgeMasks(const VPInstruction *SI) {
}
void VPPredicator::convertPhisToBlends(VPBasicBlock *VPBB) {
+ Builder.setInsertPoint(VPBB, InsertPoints[VPBB]);
SmallVector<VPPhi *> Phis;
for (VPRecipeBase &R : VPBB->phis())
Phis.push_back(cast<VPPhi>(&R));
@@ -276,10 +283,8 @@ void VPlanTransforms::introduceMasksAndLinearize(VPlan &Plan) {
// Introduce the mask for VPBB, which may introduce needed edge masks, and
// convert all phi recipes of VPBB to blend recipes unless VPBB is the
// header.
- if (VPBB != Header) {
+ if (VPBB != Header)
Predicator.createBlockInMask(VPBB);
- Predicator.convertPhisToBlends(VPBB);
- }
VPValue *BlockMask = Predicator.getBlockInMask(VPBB);
if (!BlockMask)
@@ -292,6 +297,10 @@ void VPlanTransforms::introduceMasksAndLinearize(VPlan &Plan) {
}
}
+ for (VPBlockBase *VPB : reverse(RPOT))
+ if (VPB != Header)
+ Predicator.convertPhisToBlends(cast<VPBasicBlock>(VPB));
+
// Linearize the blocks of the loop into one serial chain.
VPBlockBase *PrevVPBB = nullptr;
for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(RPOT)) {
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index e8000cc8a25c3..6f33d05b044e6 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -665,6 +665,8 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): merge.a
; CHECK-EMPTY:
; CHECK-NEXT: merge.a:
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = not ir<%c0>
+; CHECK-NEXT: BLEND ir<%blend.a> = ir<%v1>/ir<%c0> ir<%v1>/vp<[[VP5]]>
; CHECK-NEXT: EMIT ir<%d0> = icmp sgt ir<%iv>, ir<0>
; CHECK-NEXT: Successor(s): if.b
; CHECK-EMPTY:
@@ -673,14 +675,16 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): if.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: if.b.inner:
-; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
; CHECK-NEXT: Successor(s): merge.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: merge.b.inner:
; CHECK-NEXT: Successor(s): merge.b
; CHECK-EMPTY:
; CHECK-NEXT: merge.b:
-; CHECK-NEXT: EMIT ir<%sum> = add ir<%v1>, ir<%v2>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = not ir<%d0>
+; CHECK-NEXT: BLEND ir<%blend.b> = ir<%v2>/ir<%d0> ir<%v2>/vp<[[VP7]]>
+; CHECK-NEXT: EMIT ir<%sum> = add ir<%blend.a>, ir<%blend.b>
; CHECK-NEXT: EMIT store ir<%sum>, ir<%gep>
; CHECK-NEXT: Successor(s): loop.latch
; CHECK-EMPTY:
>From 6b109a45eb763a1d6398aae132563ea14593c733 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Fri, 5 Jun 2026 18:31:56 +0800
Subject: [PATCH 3/8] Assert InsertPoints isn't clobbered
---
llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp | 6 ++++--
1 file changed, 4 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index b01fef556f8ed..d8f042950e73a 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -139,8 +139,10 @@ void VPPredicator::createBlockInMask(VPBasicBlock *VPBB) {
Builder.setInsertPoint(VPBB, VPBB->getFirstNonPhi());
// Keep track of where in VPBB we are inserting the masks into.
- scope_exit UpdateInsertPoint(
- [this, &VPBB]() { InsertPoints[VPBB] = Builder.getInsertPoint(); });
+ scope_exit UpdateInsertPoint([this, &VPBB]() {
+ assert(!InsertPoints.contains(VPBB) && "InsertPoint clobbered?");
+ InsertPoints[VPBB] = Builder.getInsertPoint();
+ });
// Reuse the mask of the immediate dominator if the VPBB post-dominates the
// immediate dominator.
>From 73427ee2b2dec03dd65f9f6c65e4bb0d31a969f6 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Mon, 8 Jun 2026 15:13:55 +0800
Subject: [PATCH 4/8] Remove InsertPoints map
---
.../Transforms/Vectorize/VPlanPredicator.cpp | 18 ++++++++++--------
1 file changed, 10 insertions(+), 8 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index d8f042950e73a..2878a619119de 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -69,7 +69,14 @@ class VPPredicator {
return EdgeMaskCache[{Src, Dst}] = Mask;
}
- DenseMap<const VPBasicBlock *, VPBasicBlock::iterator> InsertPoints;
+ /// Returns where to insert new masks in \p VPBB.
+ VPBasicBlock::iterator getMaskInsertPoint(VPBasicBlock *VPBB) {
+ if (VPValue *Mask = getBlockInMask(VPBB))
+ if (VPRecipeBase *MaskR = Mask->getDefiningRecipe())
+ if (MaskR->getParent() == VPBB) // In-mask may be the IDom's.
+ return std::next(MaskR->getIterator());
+ return VPBB->getFirstNonPhi();
+ }
public:
VPPredicator(VPlan &Plan) : VPDT(Plan), VPPDT(Plan) {}
@@ -138,12 +145,6 @@ void VPPredicator::createBlockInMask(VPBasicBlock *VPBB) {
// Start inserting after the block's phis, which be replaced by blends later.
Builder.setInsertPoint(VPBB, VPBB->getFirstNonPhi());
- // Keep track of where in VPBB we are inserting the masks into.
- scope_exit UpdateInsertPoint([this, &VPBB]() {
- assert(!InsertPoints.contains(VPBB) && "InsertPoint clobbered?");
- InsertPoints[VPBB] = Builder.getInsertPoint();
- });
-
// Reuse the mask of the immediate dominator if the VPBB post-dominates the
// immediate dominator.
auto *IDom = VPDT.getNode(VPBB)->getIDom();
@@ -233,7 +234,8 @@ void VPPredicator::createSwitchEdgeMasks(const VPInstruction *SI) {
}
void VPPredicator::convertPhisToBlends(VPBasicBlock *VPBB) {
- Builder.setInsertPoint(VPBB, InsertPoints[VPBB]);
+ Builder.setInsertPoint(VPBB, getMaskInsertPoint(VPBB));
+
SmallVector<VPPhi *> Phis;
for (VPRecipeBase &R : VPBB->phis())
Phis.push_back(cast<VPPhi>(&R));
>From f07305e0aac917d07f54924fde917ce4877321e2 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Tue, 2 Jun 2026 18:01:09 +0800
Subject: [PATCH 5/8] Precommit test case
---
.../VPlan/predicator-early-exit.ll | 337 ++++++++++++++++++
1 file changed, 337 insertions(+)
create mode 100644 llvm/test/Transforms/LoopVectorize/VPlan/predicator-early-exit.ll
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator-early-exit.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator-early-exit.ll
new file mode 100644
index 0000000000000..674ab702be327
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator-early-exit.ll
@@ -0,0 +1,337 @@
+; NOTE: Assertions have been autogenerated by utils/update_analyze_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -disable-output < %s -p loop-vectorize -vplan-print-after=introduceMasksAndLinearize -vplan-print-vector-region-scope 2>&1 | FileCheck %s
+
+; Test various CFGs generated by early exits
+
+; vector.body
+; / \
+; / \
+; exiting1 \
+; / \ \
+; / \ |
+; / join1 |
+; / / \ |
+; / exiting2 \ |
+; \ | \ \ |
+; \ | join2
+; \ | /
+; \ | /
+; latch
+define void @multi_early_exit_predicated_nested(ptr %p1, ptr %p2, i1 %c1, i1 %c2, i1 %ee1, i1 %ee2, i32 %n) {
+; CHECK-LABEL: VPlan for loop in 'multi_early_exit_predicated_nested'
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK-NEXT: Successor(s): exiting1
+; CHECK-EMPTY:
+; CHECK-NEXT: exiting1:
+; CHECK-NEXT: Successor(s): join1
+; CHECK-EMPTY:
+; CHECK-NEXT: join1:
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = not ir<%ee1>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = logical-and ir<%c1>, vp<[[VP4]]>
+; CHECK-NEXT: Successor(s): exiting2
+; CHECK-EMPTY:
+; CHECK-NEXT: exiting2:
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and vp<[[VP5]]>, ir<%c2>
+; CHECK-NEXT: Successor(s): join2
+; CHECK-EMPTY:
+; CHECK-NEXT: join2:
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = not ir<%ee2>
+; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = logical-and vp<[[VP6]]>, vp<[[VP7]]>
+; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = not ir<%c2>
+; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = logical-and vp<[[VP5]]>, vp<[[VP9]]>
+; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = or vp<[[VP8]]>, vp<[[VP10]]>
+; CHECK-NEXT: EMIT vp<[[VP12:%[0-9]+]]> = not ir<%c1>
+; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = or vp<[[VP11]]>, vp<[[VP12]]>
+; CHECK-NEXT: BLEND ir<%phi1.join2> = ir<1>/vp<[[VP8]]> ir<1>/vp<[[VP10]]> ir<0>/vp<[[VP12]]>
+; CHECK-NEXT: BLEND ir<%phi2.join2> = ir<1>/vp<[[VP8]]> ir<0>/vp<[[VP10]]> ir<0>/vp<[[VP12]]>
+; CHECK-NEXT: Successor(s): latch
+; CHECK-EMPTY:
+; CHECK-NEXT: latch:
+; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = logical-and vp<[[VP6]]>, ir<%ee2>
+; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = logical-and ir<%c1>, ir<%ee1>
+; CHECK-NEXT: BLEND ir<%phi1> = ir<%phi1.join2>/vp<[[VP13]]> ir<1>/vp<[[VP14]]> ir<1>/vp<[[VP15]]>
+; CHECK-NEXT: BLEND ir<%phi2> = ir<%phi2.join2>/vp<[[VP13]]> ir<1>/vp<[[VP14]]> ir<poison>/vp<[[VP15]]>
+; CHECK-NEXT: EMIT ir<%gep1> = getelementptr ir<%p1>, ir<%iv>
+; CHECK-NEXT: EMIT store ir<%phi1>, ir<%gep1>
+; CHECK-NEXT: EMIT ir<%gep2> = getelementptr ir<%p2>, ir<%iv>
+; CHECK-NEXT: EMIT store ir<%phi2>, ir<%gep2>
+; CHECK-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; CHECK-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<%n>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK-NEXT: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [0, %entry], [%iv.next, %latch]
+ br i1 %c1, label %exiting1, label %join2
+
+exiting1:
+ br i1 %ee1, label %latch, label %join1
+
+join1:
+ br i1 %c2, label %exiting2, label %join2
+
+exiting2:
+ br i1 %ee2, label %latch, label %join2
+
+join2:
+ %phi1.join2 = phi i32 [1, %exiting2], [1, %join1], [0, %loop]
+ %phi2.join2 = phi i32 [1, %exiting2], [0, %join1], [0, %loop]
+ br label %latch
+
+latch:
+ %phi1 = phi i32 [1, %exiting1], [1, %exiting2], [%phi1.join2, %join2]
+ %phi2 = phi i32 [poison, %exiting1], [1, %exiting2], [%phi2.join2, %join2]
+ %gep1 = getelementptr i32, ptr %p1, i32 %iv
+ store i32 %phi1, ptr %gep1
+ %gep2 = getelementptr i32, ptr %p2, i32 %iv
+ store i32 %phi2, ptr %gep2
+ %iv.next = add i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, %n
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+; vector.body
+; / \
+; / \
+; exiting1 /
+; / \ /
+; / \ /
+; / join1
+; / / \
+; / exiting2 \
+; \ | \ \
+; \ | join2
+; \ | /
+; \ | /
+; latch
+define void @multi_early_exit_predicated_not_nested(ptr %p1, ptr %p2, i1 %c1, i1 %c2, i1 %ee1, i1 %ee2, i32 %n) {
+; CHECK-LABEL: VPlan for loop in 'multi_early_exit_predicated_not_nested'
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK-NEXT: Successor(s): exiting1
+; CHECK-EMPTY:
+; CHECK-NEXT: exiting1:
+; CHECK-NEXT: Successor(s): join1
+; CHECK-EMPTY:
+; CHECK-NEXT: join1:
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = not ir<%ee1>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = logical-and ir<%c1>, vp<[[VP4]]>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = not ir<%c1>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = or vp<[[VP5]]>, vp<[[VP6]]>
+; CHECK-NEXT: BLEND ir<%phi.join1> = ir<1>/vp<[[VP5]]> ir<0>/vp<[[VP6]]>
+; CHECK-NEXT: Successor(s): exiting2
+; CHECK-EMPTY:
+; CHECK-NEXT: exiting2:
+; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = logical-and vp<[[VP7]]>, ir<%c2>
+; CHECK-NEXT: Successor(s): join2
+; CHECK-EMPTY:
+; CHECK-NEXT: join2:
+; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = not ir<%ee2>
+; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = logical-and vp<[[VP8]]>, vp<[[VP9]]>
+; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = not ir<%c2>
+; CHECK-NEXT: EMIT vp<[[VP12:%[0-9]+]]> = logical-and vp<[[VP7]]>, vp<[[VP11]]>
+; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = or vp<[[VP10]]>, vp<[[VP12]]>
+; CHECK-NEXT: BLEND ir<%phi.join2> = ir<1>/vp<[[VP10]]> ir<0>/vp<[[VP12]]>
+; CHECK-NEXT: Successor(s): latch
+; CHECK-EMPTY:
+; CHECK-NEXT: latch:
+; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = logical-and vp<[[VP8]]>, ir<%ee2>
+; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = logical-and ir<%c1>, ir<%ee1>
+; CHECK-NEXT: BLEND ir<%phi1> = ir<%phi.join1>/vp<[[VP13]]> ir<%phi.join1>/vp<[[VP14]]> ir<1>/vp<[[VP15]]>
+; CHECK-NEXT: BLEND ir<%phi2> = ir<%phi.join2>/vp<[[VP13]]> ir<1>/vp<[[VP14]]> ir<poison>/vp<[[VP15]]>
+; CHECK-NEXT: EMIT ir<%gep1> = getelementptr ir<%p1>, ir<%iv>
+; CHECK-NEXT: EMIT store ir<%phi1>, ir<%gep1>
+; CHECK-NEXT: EMIT ir<%gep2> = getelementptr ir<%p2>, ir<%iv>
+; CHECK-NEXT: EMIT store ir<%phi2>, ir<%gep2>
+; CHECK-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; CHECK-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<%n>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK-NEXT: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [0, %entry], [%iv.next, %latch]
+ br i1 %c1, label %exiting1, label %join1
+
+exiting1:
+ br i1 %ee1, label %latch, label %join1
+
+join1:
+ %phi.join1 = phi i32 [1, %exiting1], [0, %loop]
+ br i1 %c2, label %exiting2, label %join2
+
+exiting2:
+ br i1 %ee2, label %latch, label %join2
+
+join2:
+ %phi.join2 = phi i32 [1, %exiting2], [0, %join1]
+ br label %latch
+
+latch:
+ %phi1 = phi i32 [1, %exiting1], [%phi.join1, %exiting2], [%phi.join1, %join2]
+ %phi2 = phi i32 [poison, %exiting1], [1, %exiting2], [%phi.join2, %join2]
+ %gep1 = getelementptr i32, ptr %p1, i32 %iv
+ store i32 %phi1, ptr %gep1
+ %gep2 = getelementptr i32, ptr %p2, i32 %iv
+ store i32 %phi2, ptr %gep2
+ %iv.next = add i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, %n
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
+
+; vector.body
+; / \
+; / \
+; exiting1 exiting2
+; / \ / \
+; / \ / \
+; / join1 \
+; / / \ \
+; / exiting3 exiting4 /
+; \ \ \ / / /
+; \ \ join2 / /
+; \ \ | / /
+; +---latch-----+
+define void @four_exits_2x2_diamond(ptr %p1, ptr %p2, ptr %p3, ptr %p4, i1 %c1, i1 %c2, i1 %ee1, i1 %ee2, i1 %ee3, i1 %ee4, i32 %n) {
+; CHECK-LABEL: VPlan for loop in 'four_exits_2x2_diamond'
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK-NEXT: Successor(s): exiting2
+; CHECK-EMPTY:
+; CHECK-NEXT: exiting2:
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = not ir<%c1>
+; CHECK-NEXT: Successor(s): exiting1
+; CHECK-EMPTY:
+; CHECK-NEXT: exiting1:
+; CHECK-NEXT: Successor(s): join1
+; CHECK-EMPTY:
+; CHECK-NEXT: join1:
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = not ir<%ee2>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and vp<[[VP4]]>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = not ir<%ee1>
+; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = logical-and ir<%c1>, vp<[[VP7]]>
+; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = or vp<[[VP6]]>, vp<[[VP8]]>
+; CHECK-NEXT: BLEND ir<%phi1.join1> = ir<0>/vp<[[VP6]]> ir<1>/vp<[[VP8]]>
+; CHECK-NEXT: BLEND ir<%phi2.join1> = ir<1>/vp<[[VP6]]> ir<0>/vp<[[VP8]]>
+; CHECK-NEXT: Successor(s): exiting4
+; CHECK-EMPTY:
+; CHECK-NEXT: exiting4:
+; CHECK-NEXT: EMIT vp<[[VP10:%[0-9]+]]> = not ir<%c2>
+; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = logical-and vp<[[VP9]]>, vp<[[VP10]]>
+; CHECK-NEXT: Successor(s): exiting3
+; CHECK-EMPTY:
+; CHECK-NEXT: exiting3:
+; CHECK-NEXT: EMIT vp<[[VP12:%[0-9]+]]> = logical-and vp<[[VP9]]>, ir<%c2>
+; CHECK-NEXT: Successor(s): join2
+; CHECK-EMPTY:
+; CHECK-NEXT: join2:
+; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = not ir<%ee4>
+; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = logical-and vp<[[VP11]]>, vp<[[VP13]]>
+; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = not ir<%ee3>
+; CHECK-NEXT: EMIT vp<[[VP16:%[0-9]+]]> = logical-and vp<[[VP12]]>, vp<[[VP15]]>
+; CHECK-NEXT: EMIT vp<[[VP17:%[0-9]+]]> = or vp<[[VP14]]>, vp<[[VP16]]>
+; CHECK-NEXT: BLEND ir<%phi3.join2> = ir<0>/vp<[[VP14]]> ir<1>/vp<[[VP16]]>
+; CHECK-NEXT: BLEND ir<%phi4.join2> = ir<1>/vp<[[VP14]]> ir<0>/vp<[[VP16]]>
+; CHECK-NEXT: Successor(s): latch
+; CHECK-EMPTY:
+; CHECK-NEXT: latch:
+; CHECK-NEXT: EMIT vp<[[VP18:%[0-9]+]]> = logical-and vp<[[VP11]]>, ir<%ee4>
+; CHECK-NEXT: EMIT vp<[[VP19:%[0-9]+]]> = logical-and vp<[[VP12]]>, ir<%ee3>
+; CHECK-NEXT: EMIT vp<[[VP20:%[0-9]+]]> = logical-and vp<[[VP4]]>, ir<%ee2>
+; CHECK-NEXT: EMIT vp<[[VP21:%[0-9]+]]> = logical-and ir<%c1>, ir<%ee1>
+; CHECK-NEXT: BLEND ir<%phi1> = ir<%phi1.join1>/vp<[[VP17]]> ir<%phi1.join1>/vp<[[VP18]]> ir<%phi1.join1>/vp<[[VP19]]> ir<0>/vp<[[VP20]]> ir<1>/vp<[[VP21]]>
+; CHECK-NEXT: BLEND ir<%phi2> = ir<%phi2.join1>/vp<[[VP17]]> ir<%phi2.join1>/vp<[[VP18]]> ir<%phi2.join1>/vp<[[VP19]]> ir<1>/vp<[[VP20]]> ir<poison>/vp<[[VP21]]>
+; CHECK-NEXT: BLEND ir<%phi3> = ir<%phi3.join2>/vp<[[VP17]]> ir<0>/vp<[[VP18]]> ir<1>/vp<[[VP19]]> ir<poison>/vp<[[VP20]]> ir<poison>/vp<[[VP21]]>
+; CHECK-NEXT: BLEND ir<%phi4> = ir<%phi4.join2>/vp<[[VP17]]> ir<1>/vp<[[VP18]]> ir<poison>/vp<[[VP19]]> ir<poison>/vp<[[VP20]]> ir<poison>/vp<[[VP21]]>
+; CHECK-NEXT: EMIT ir<%gep1> = getelementptr ir<%p1>, ir<%iv>
+; CHECK-NEXT: EMIT store ir<%phi1>, ir<%gep1>
+; CHECK-NEXT: EMIT ir<%gep2> = getelementptr ir<%p2>, ir<%iv>
+; CHECK-NEXT: EMIT store ir<%phi2>, ir<%gep2>
+; CHECK-NEXT: EMIT ir<%gep3> = getelementptr ir<%p3>, ir<%iv>
+; CHECK-NEXT: EMIT store ir<%phi3>, ir<%gep3>
+; CHECK-NEXT: EMIT ir<%gep4> = getelementptr ir<%p4>, ir<%iv>
+; CHECK-NEXT: EMIT store ir<%phi4>, ir<%gep4>
+; CHECK-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; CHECK-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<%n>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK-NEXT: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [0, %entry], [%iv.next, %latch]
+ br i1 %c1, label %exiting1, label %exiting2
+
+exiting1:
+ br i1 %ee1, label %latch, label %join1
+
+exiting2:
+ br i1 %ee2, label %latch, label %join1
+
+join1:
+ %phi1.join1 = phi i32 [1, %exiting1], [0, %exiting2]
+ %phi2.join1 = phi i32 [0, %exiting1], [1, %exiting2]
+ br i1 %c2, label %exiting3, label %exiting4
+
+exiting3:
+ br i1 %ee3, label %latch, label %join2
+
+exiting4:
+ br i1 %ee4, label %latch, label %join2
+
+join2:
+ %phi3.join2 = phi i32 [1, %exiting3], [0, %exiting4]
+ %phi4.join2 = phi i32 [0, %exiting3], [1, %exiting4]
+ br label %latch
+
+latch:
+ %phi1 = phi i32 [1, %exiting1], [0, %exiting2], [%phi1.join1, %exiting3], [%phi1.join1, %exiting4], [%phi1.join1, %join2]
+ %phi2 = phi i32 [poison, %exiting1], [1, %exiting2], [%phi2.join1, %exiting3], [%phi2.join1, %exiting4], [%phi2.join1, %join2]
+ %phi3 = phi i32 [poison, %exiting1], [poison, %exiting2], [1, %exiting3], [0, %exiting4], [%phi3.join2, %join2]
+ %phi4 = phi i32 [poison, %exiting1], [poison, %exiting2], [poison, %exiting3], [1, %exiting4], [%phi4.join2, %join2]
+ %gep1 = getelementptr i32, ptr %p1, i32 %iv
+ store i32 %phi1, ptr %gep1
+ %gep2 = getelementptr i32, ptr %p2, i32 %iv
+ store i32 %phi2, ptr %gep2
+ %gep3 = getelementptr i32, ptr %p3, i32 %iv
+ store i32 %phi3, ptr %gep3
+ %gep4 = getelementptr i32, ptr %p4, i32 %iv
+ store i32 %phi4, ptr %gep4
+ %iv.next = add i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, %n
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
>From 0ac59569871203c619f838efcc97ce6081f2adb6 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Wed, 3 Jun 2026 23:55:02 +0800
Subject: [PATCH 6/8] Simplify blend masks
---
.../include/llvm/Analysis/DominanceFrontier.h | 1 +
.../Transforms/Vectorize/VPlanDominatorTree.h | 8 +
.../Transforms/Vectorize/VPlanPredicator.cpp | 147 +++++++++++++++++-
.../VPlan/predicator-early-exit.ll | 30 ++--
.../LoopVectorize/VPlan/predicator.ll | 10 +-
.../LoopVectorize/predicate-switch.ll | 9 +-
.../LoopVectorize/reduction-inloop-pred.ll | 10 +-
.../LoopVectorize/reduction-inloop.ll | 21 +--
.../Transforms/LoopVectorize/reduction.ll | 10 +-
9 files changed, 191 insertions(+), 55 deletions(-)
diff --git a/llvm/include/llvm/Analysis/DominanceFrontier.h b/llvm/include/llvm/Analysis/DominanceFrontier.h
index fd38891e901e3..4a8ab96cf71a7 100644
--- a/llvm/include/llvm/Analysis/DominanceFrontier.h
+++ b/llvm/include/llvm/Analysis/DominanceFrontier.h
@@ -78,6 +78,7 @@ class DominanceFrontierBase {
const_iterator end() const { return Frontiers.end(); }
iterator find(BlockT *B) { return Frontiers.find(B); }
const_iterator find(BlockT *B) const { return Frontiers.find(B); }
+ const_iterator find(const BlockT *B) const { return Frontiers.find(B); }
/// print - Convert to human readable form
///
diff --git a/llvm/lib/Transforms/Vectorize/VPlanDominatorTree.h b/llvm/lib/Transforms/Vectorize/VPlanDominatorTree.h
index 2864670f44913..1ad522880c709 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanDominatorTree.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanDominatorTree.h
@@ -18,6 +18,8 @@
#include "VPlan.h"
#include "VPlanCFG.h"
#include "llvm/ADT/GraphTraits.h"
+#include "llvm/Analysis/DominanceFrontier.h"
+#include "llvm/Analysis/DominanceFrontierImpl.h"
#include "llvm/IR/Dominators.h"
#include "llvm/Support/GenericDomTree.h"
#include "llvm/Support/GenericDomTreeConstruction.h"
@@ -67,5 +69,11 @@ template <>
struct GraphTraits<const VPDomTreeNode *>
: public DomTreeGraphTraitsBase<const VPDomTreeNode,
VPDomTreeNode::const_iterator> {};
+
+class VPPostDominanceFrontier
+ : public DominanceFrontierBase<VPBlockBase, true> {
+public:
+ explicit VPPostDominanceFrontier(const DomTreeT &VPDT) { analyze(VPDT); }
+};
} // namespace llvm
#endif // LLVM_TRANSFORMS_VECTORIZE_VPLANDOMINATORTREE_H
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 2878a619119de..5f74b1cae0630 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -34,6 +34,9 @@ class VPPredicator {
/// Post-dominator tree for the VPlan.
VPPostDominatorTree VPPDT;
+ /// Post-dominator frontier for the VPlan.
+ VPPostDominanceFrontier VPPDF;
+
/// When we if-convert we need to create edge masks. We have to cache values
/// so that we don't end up with exponential recursion/IR.
using EdgeMaskCacheTy =
@@ -78,8 +81,17 @@ class VPPredicator {
return VPBB->getFirstNonPhi();
}
+ using EdgeTy = std::pair<const VPBasicBlock *, const VPBasicBlock *>;
+
+ /// Compute the "furthest up" set of edges for each incoming value of \Phi.
+ MapVector<EdgeTy, VPValue *> computeBlendEdges(VPPhi *Phi);
+
+ /// Given a set of \p Edges that lead to \p VPBB, return the OR of all edges
+ /// or an equivalent block in-mask.
+ VPValue *createMaskDisjunction(ArrayRef<EdgeTy> Edges, VPBasicBlock *VPBB);
+
public:
- VPPredicator(VPlan &Plan) : VPDT(Plan), VPPDT(Plan) {}
+ VPPredicator(VPlan &Plan) : VPDT(Plan), VPPDT(Plan), VPPDF(VPPDT) {}
/// Returns the *entry* mask for \p VPBB.
VPValue *getBlockInMask(const VPBasicBlock *VPBB) const {
@@ -233,6 +245,115 @@ void VPPredicator::createSwitchEdgeMasks(const VPInstruction *SI) {
setEdgeMask(Src, DefaultDst, DefaultMask);
}
+// Compute the "furthest up" set of edges for each incoming value of a phi.
+//
+// Start by keeping track of what edges lead to which value. Then see if any
+// node has the same value for all outgoing edges. If so then propagate that
+// value up to every node it postdominates.
+MapVector<VPPredicator::EdgeTy, VPValue *>
+VPPredicator::computeBlendEdges(VPPhi *Phi) {
+ MapVector<EdgeTy, VPValue *> Edges;
+
+ // Mark the given edge as providing the value \p V.
+ auto AddEdge = [&Edges](const VPBlockBase *From, const VPBlockBase *To,
+ VPValue *V) {
+ EdgeTy Edge = {cast<VPBasicBlock>(From), cast<VPBasicBlock>(To)};
+ assert((!Edges.contains(Edge) || Edges.lookup(Edge) == V) &&
+ "Clobbering an edge?");
+ Edges[Edge] = V;
+ };
+
+ for (auto [InVal, InVPBB] : Phi->incoming_values_and_blocks())
+ AddEdge(InVPBB, Phi->getParent(), InVal);
+
+ // The root phi must postdominate every incoming block. Also don't touch
+ // phis in a reduction chain since they need to be in a specific structure
+ // for handle*Reductions.
+ for (auto [InVal, InVPBB] : Phi->incoming_values_and_blocks())
+ if (!VPPDT.dominates(Phi->getParent(), InVPBB) ||
+ isa<VPReductionPHIRecipe>(InVal))
+ return Edges;
+
+ // Given a list of edges, check if they all have the same value and return it.
+ auto GetAllEqual = [&Edges](ArrayRef<EdgeTy> OutEdges) -> VPValue * {
+ VPValue *Common = nullptr;
+ for (EdgeTy E : OutEdges) {
+ VPValue *V = Edges.lookup(E);
+ if (!V)
+ return nullptr;
+ if (match(V, m_Poison()))
+ continue;
+ if (!Common)
+ Common = V;
+ else if (Common != V)
+ return nullptr;
+ }
+ return Common;
+ };
+
+ SetVector<const VPBlockBase *> Worklist(from_range, Phi->incoming_blocks());
+ while (!Worklist.empty()) {
+ auto *VPBB = cast<VPBasicBlock>(Worklist.pop_back_val());
+
+ // Check that all outgoing edges from VPBB have the same value.
+ SmallVector<EdgeTy> OutEdges;
+ for (const VPBlockBase *Succ : VPBB->getSuccessors())
+ OutEdges.emplace_back(VPBB, cast<VPBasicBlock>(Succ));
+ VPValue *Common = GetAllEqual(OutEdges);
+ if (!Common)
+ continue;
+
+ // They have the same value: we can move the edges up
+ for (EdgeTy Edge : OutEdges)
+ Edges.erase(Edge);
+
+ // Peek through phis that are postdominated by VPBB
+ if (auto *Phi = dyn_cast<VPPhi>(Common))
+ if (VPPDT.dominates(VPBB, Phi->getParent())) {
+ for (auto [InV, InVPBB] : Phi->incoming_values_and_blocks()) {
+ AddEdge(InVPBB, Phi->getParent(), InV);
+ Worklist.insert(InVPBB);
+ }
+ continue;
+ }
+
+ // Iterate up through the post dominance frontier
+ for (const VPBlockBase *Frontier : VPPDF.find(VPBB)->second) {
+ for (const VPBlockBase *FrontierSucc : Frontier->getSuccessors())
+ if (VPPDT.dominates(VPBB, FrontierSucc))
+ AddEdge(Frontier, FrontierSucc, Common);
+ Worklist.insert(cast<VPBasicBlock>(Frontier));
+ }
+ }
+
+ return Edges;
+}
+
+VPValue *VPPredicator::createMaskDisjunction(ArrayRef<EdgeTy> Edges,
+ VPBasicBlock *VPBB) {
+ auto Dsts = map_range(Edges, [](auto E) { return E.second; });
+ const VPBasicBlock *PostDom = *Dsts.begin();
+ for (const VPBasicBlock *VPBB : drop_begin(Dsts))
+ PostDom =
+ cast<VPBasicBlock>(VPPDT.findNearestCommonDominator(PostDom, VPBB));
+ assert(VPPDT.dominates(VPBB, PostDom) && "Edges don't postdominate VPBB");
+ if (PostDom != VPBB)
+ return getBlockInMask(PostDom);
+
+ VPValue *Mask = nullptr;
+ for (auto [Src, ConstDst] : Edges) {
+ auto *Dst = const_cast<VPBasicBlock *>(ConstDst);
+ VPValue *EdgeMask;
+ {
+ VPBuilder::InsertPointGuard Guard(Builder);
+ Builder.setInsertPoint(Dst, getMaskInsertPoint(Dst));
+ EdgeMask = createEdgeMask(Src, Dst);
+ }
+ Mask = Mask ? Builder.createOr(Mask, EdgeMask) : EdgeMask;
+ }
+ return Mask;
+}
+
void VPPredicator::convertPhisToBlends(VPBasicBlock *VPBB) {
Builder.setInsertPoint(VPBB, getMaskInsertPoint(VPBB));
@@ -256,10 +377,30 @@ void VPPredicator::convertPhisToBlends(VPBasicBlock *VPBB) {
continue;
}
+ MapVector<VPValue *, SmallVector<EdgeTy>> InValEdgesMap;
+ for (auto [Edge, Val] : computeBlendEdges(PhiR))
+ InValEdgesMap[Val].push_back(Edge);
+ auto InValEdges = InValEdgesMap.takeVector();
+
+ if (InValEdges.size() == 1) {
+ PhiR->replaceAllUsesWith(InValEdges[0].first);
+ PhiR->eraseFromParent();
+ continue;
+ }
+
+ // Sort the incoming value order to match PhiR as much as possible.
+ llvm::stable_sort(InValEdges, [&PhiR](auto &L, auto &R) {
+ auto InVs = PhiR->incoming_values();
+ return std::distance(InVs.begin(), find(InVs, L.first)) <
+ std::distance(InVs.begin(), find(InVs, R.first));
+ });
+
SmallVector<VPValue *, 2> OperandsWithMask;
- for (const auto &[InVPV, InVPBB] : PhiR->incoming_values_and_blocks()) {
+ for (const auto &[InVPV, Edges] : InValEdges) {
+ if (match(InVPV, m_Poison()))
+ continue;
OperandsWithMask.push_back(InVPV);
- OperandsWithMask.push_back(createEdgeMask(InVPBB, VPBB));
+ OperandsWithMask.push_back(createMaskDisjunction(Edges, VPBB));
}
PHINode *IRPhi = cast_or_null<PHINode>(PhiR->getUnderlyingValue());
auto *Blend =
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator-early-exit.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator-early-exit.ll
index 674ab702be327..4d388a3be84e4 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator-early-exit.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator-early-exit.ll
@@ -46,15 +46,15 @@ define void @multi_early_exit_predicated_nested(ptr %p1, ptr %p2, i1 %c1, i1 %c2
; CHECK-NEXT: EMIT vp<[[VP11:%[0-9]+]]> = or vp<[[VP8]]>, vp<[[VP10]]>
; CHECK-NEXT: EMIT vp<[[VP12:%[0-9]+]]> = not ir<%c1>
; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = or vp<[[VP11]]>, vp<[[VP12]]>
-; CHECK-NEXT: BLEND ir<%phi1.join2> = ir<1>/vp<[[VP8]]> ir<1>/vp<[[VP10]]> ir<0>/vp<[[VP12]]>
-; CHECK-NEXT: BLEND ir<%phi2.join2> = ir<1>/vp<[[VP8]]> ir<0>/vp<[[VP10]]> ir<0>/vp<[[VP12]]>
+; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = or vp<[[VP8]]>, vp<[[VP10]]>
+; CHECK-NEXT: BLEND ir<%phi1.join2> = ir<1>/vp<[[VP14]]> ir<0>/vp<[[VP12]]>
+; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = or vp<[[VP10]]>, vp<[[VP12]]>
+; CHECK-NEXT: BLEND ir<%phi2.join2> = ir<1>/vp<[[VP8]]> ir<0>/vp<[[VP15]]>
; CHECK-NEXT: Successor(s): latch
; CHECK-EMPTY:
; CHECK-NEXT: latch:
-; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = logical-and vp<[[VP6]]>, ir<%ee2>
-; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = logical-and ir<%c1>, ir<%ee1>
-; CHECK-NEXT: BLEND ir<%phi1> = ir<%phi1.join2>/vp<[[VP13]]> ir<1>/vp<[[VP14]]> ir<1>/vp<[[VP15]]>
-; CHECK-NEXT: BLEND ir<%phi2> = ir<%phi2.join2>/vp<[[VP13]]> ir<1>/vp<[[VP14]]> ir<poison>/vp<[[VP15]]>
+; CHECK-NEXT: BLEND ir<%phi1> = ir<1>/ir<%c1> ir<0>/vp<[[VP13]]>
+; CHECK-NEXT: BLEND ir<%phi2> = ir<1>/vp<[[VP6]]> ir<0>/vp<[[VP13]]>
; CHECK-NEXT: EMIT ir<%gep1> = getelementptr ir<%p1>, ir<%iv>
; CHECK-NEXT: EMIT store ir<%phi1>, ir<%gep1>
; CHECK-NEXT: EMIT ir<%gep2> = getelementptr ir<%p2>, ir<%iv>
@@ -151,10 +151,8 @@ define void @multi_early_exit_predicated_not_nested(ptr %p1, ptr %p2, i1 %c1, i1
; CHECK-NEXT: Successor(s): latch
; CHECK-EMPTY:
; CHECK-NEXT: latch:
-; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = logical-and vp<[[VP8]]>, ir<%ee2>
-; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = logical-and ir<%c1>, ir<%ee1>
-; CHECK-NEXT: BLEND ir<%phi1> = ir<%phi.join1>/vp<[[VP13]]> ir<%phi.join1>/vp<[[VP14]]> ir<1>/vp<[[VP15]]>
-; CHECK-NEXT: BLEND ir<%phi2> = ir<%phi.join2>/vp<[[VP13]]> ir<1>/vp<[[VP14]]> ir<poison>/vp<[[VP15]]>
+; CHECK-NEXT: BLEND ir<%phi1> = ir<1>/ir<%c1> ir<0>/vp<[[VP7]]>
+; CHECK-NEXT: BLEND ir<%phi2> = ir<1>/vp<[[VP8]]> ir<0>/vp<[[VP13]]>
; CHECK-NEXT: EMIT ir<%gep1> = getelementptr ir<%p1>, ir<%iv>
; CHECK-NEXT: EMIT store ir<%phi1>, ir<%gep1>
; CHECK-NEXT: EMIT ir<%gep2> = getelementptr ir<%p2>, ir<%iv>
@@ -262,14 +260,10 @@ define void @four_exits_2x2_diamond(ptr %p1, ptr %p2, ptr %p3, ptr %p4, i1 %c1,
; CHECK-NEXT: Successor(s): latch
; CHECK-EMPTY:
; CHECK-NEXT: latch:
-; CHECK-NEXT: EMIT vp<[[VP18:%[0-9]+]]> = logical-and vp<[[VP11]]>, ir<%ee4>
-; CHECK-NEXT: EMIT vp<[[VP19:%[0-9]+]]> = logical-and vp<[[VP12]]>, ir<%ee3>
-; CHECK-NEXT: EMIT vp<[[VP20:%[0-9]+]]> = logical-and vp<[[VP4]]>, ir<%ee2>
-; CHECK-NEXT: EMIT vp<[[VP21:%[0-9]+]]> = logical-and ir<%c1>, ir<%ee1>
-; CHECK-NEXT: BLEND ir<%phi1> = ir<%phi1.join1>/vp<[[VP17]]> ir<%phi1.join1>/vp<[[VP18]]> ir<%phi1.join1>/vp<[[VP19]]> ir<0>/vp<[[VP20]]> ir<1>/vp<[[VP21]]>
-; CHECK-NEXT: BLEND ir<%phi2> = ir<%phi2.join1>/vp<[[VP17]]> ir<%phi2.join1>/vp<[[VP18]]> ir<%phi2.join1>/vp<[[VP19]]> ir<1>/vp<[[VP20]]> ir<poison>/vp<[[VP21]]>
-; CHECK-NEXT: BLEND ir<%phi3> = ir<%phi3.join2>/vp<[[VP17]]> ir<0>/vp<[[VP18]]> ir<1>/vp<[[VP19]]> ir<poison>/vp<[[VP20]]> ir<poison>/vp<[[VP21]]>
-; CHECK-NEXT: BLEND ir<%phi4> = ir<%phi4.join2>/vp<[[VP17]]> ir<1>/vp<[[VP18]]> ir<poison>/vp<[[VP19]]> ir<poison>/vp<[[VP20]]> ir<poison>/vp<[[VP21]]>
+; CHECK-NEXT: BLEND ir<%phi1> = ir<0>/vp<[[VP4]]> ir<1>/ir<%c1>
+; CHECK-NEXT: BLEND ir<%phi2> = ir<1>/vp<[[VP4]]> ir<0>/ir<%c1>
+; CHECK-NEXT: BLEND ir<%phi3> = ir<0>/vp<[[VP11]]> ir<1>/vp<[[VP12]]>
+; CHECK-NEXT: BLEND ir<%phi4> = ir<1>/vp<[[VP11]]> ir<0>/vp<[[VP12]]>
; CHECK-NEXT: EMIT ir<%gep1> = getelementptr ir<%p1>, ir<%iv>
; CHECK-NEXT: EMIT store ir<%phi1>, ir<%gep1>
; CHECK-NEXT: EMIT ir<%gep2> = getelementptr ir<%p2>, ir<%iv>
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index 6f33d05b044e6..487d67f41774e 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -309,7 +309,7 @@ define void @switch(ptr %a) {
; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = not vp<[[VP12]]>
; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = logical-and ir<%c0>, vp<[[VP13]]>
; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = or vp<[[VP5]]>, vp<[[VP11]]>
-; CHECK-NEXT: BLEND ir<%phi3> = ir<%add2>/vp<[[VP5]]> ir<%add1>/vp<[[VP11]]> ir<%add1>/vp<[[VP11]]>
+; CHECK-NEXT: BLEND ir<%phi3> = ir<%add2>/vp<[[VP5]]> ir<%add1>/vp<[[VP11]]>
; CHECK-NEXT: EMIT ir<%add3> = add ir<%phi3>, ir<3>, vp<[[VP15]]>
; CHECK-NEXT: Successor(s): bb4
; CHECK-EMPTY:
@@ -665,8 +665,6 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): merge.a
; CHECK-EMPTY:
; CHECK-NEXT: merge.a:
-; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = not ir<%c0>
-; CHECK-NEXT: BLEND ir<%blend.a> = ir<%v1>/ir<%c0> ir<%v1>/vp<[[VP5]]>
; CHECK-NEXT: EMIT ir<%d0> = icmp sgt ir<%iv>, ir<0>
; CHECK-NEXT: Successor(s): if.b
; CHECK-EMPTY:
@@ -675,16 +673,14 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): if.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: if.b.inner:
-; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
; CHECK-NEXT: Successor(s): merge.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: merge.b.inner:
; CHECK-NEXT: Successor(s): merge.b
; CHECK-EMPTY:
; CHECK-NEXT: merge.b:
-; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = not ir<%d0>
-; CHECK-NEXT: BLEND ir<%blend.b> = ir<%v2>/ir<%d0> ir<%v2>/vp<[[VP7]]>
-; CHECK-NEXT: EMIT ir<%sum> = add ir<%blend.a>, ir<%blend.b>
+; CHECK-NEXT: EMIT ir<%sum> = add ir<%v1>, ir<%v2>
; CHECK-NEXT: EMIT store ir<%sum>, ir<%gep>
; CHECK-NEXT: Successor(s): loop.latch
; CHECK-EMPTY:
diff --git a/llvm/test/Transforms/LoopVectorize/predicate-switch.ll b/llvm/test/Transforms/LoopVectorize/predicate-switch.ll
index 96775e0ff082e..d94b1d606e738 100644
--- a/llvm/test/Transforms/LoopVectorize/predicate-switch.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicate-switch.ll
@@ -545,8 +545,7 @@ define void @switch_unconditional_duplicate_target(ptr %start, ptr %dest) {
; IC1-NEXT: [[TMP4:%.*]] = extractelement <2 x ptr> [[TMP0]], i64 0
; IC1-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META6:![0-9]+]]
; IC1-NEXT: [[TMP5:%.*]] = icmp ult <2 x i32> [[WIDE_LOAD]], splat (i32 10)
-; IC1-NEXT: [[PREDPHI:%.*]] = select <2 x i1> [[TMP5]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP0]]
-; IC1-NEXT: [[PREDPHI2:%.*]] = select <2 x i1> [[TMP5]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[PREDPHI]]
+; IC1-NEXT: [[PREDPHI2:%.*]] = select <2 x i1> [[TMP5]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP0]]
; IC1-NEXT: [[TMP1:%.*]] = extractelement <2 x ptr> [[PREDPHI2]], i64 0
; IC1-NEXT: [[TMP2:%.*]] = extractelement <2 x ptr> [[PREDPHI2]], i64 1
; IC1-NEXT: store i32 0, ptr [[TMP1]], align 4
@@ -605,12 +604,10 @@ define void @switch_unconditional_duplicate_target(ptr %start, ptr %dest) {
; IC2-NEXT: [[WIDE_LOAD2:%.*]] = load <2 x i32>, ptr [[TMP8]], align 4, !alias.scope [[META6]]
; IC2-NEXT: [[TMP9:%.*]] = icmp ult <2 x i32> [[WIDE_LOAD]], splat (i32 10)
; IC2-NEXT: [[TMP10:%.*]] = icmp ult <2 x i32> [[WIDE_LOAD2]], splat (i32 10)
-; IC2-NEXT: [[PREDPHI:%.*]] = select <2 x i1> [[TMP9]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP0]]
-; IC2-NEXT: [[PREDPHI2:%.*]] = select <2 x i1> [[TMP9]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[PREDPHI]]
+; IC2-NEXT: [[PREDPHI2:%.*]] = select <2 x i1> [[TMP9]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP0]]
; IC2-NEXT: [[TMP2:%.*]] = extractelement <2 x ptr> [[PREDPHI2]], i64 0
; IC2-NEXT: [[TMP3:%.*]] = extractelement <2 x ptr> [[PREDPHI2]], i64 1
-; IC2-NEXT: [[PREDPHI5:%.*]] = select <2 x i1> [[TMP10]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP1]]
-; IC2-NEXT: [[PREDPHI4:%.*]] = select <2 x i1> [[TMP10]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[PREDPHI5]]
+; IC2-NEXT: [[PREDPHI4:%.*]] = select <2 x i1> [[TMP10]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP1]]
; IC2-NEXT: [[TMP4:%.*]] = extractelement <2 x ptr> [[PREDPHI4]], i64 0
; IC2-NEXT: [[TMP5:%.*]] = extractelement <2 x ptr> [[PREDPHI4]], i64 1
; IC2-NEXT: store i32 0, ptr [[TMP2]], align 4
diff --git a/llvm/test/Transforms/LoopVectorize/reduction-inloop-pred.ll b/llvm/test/Transforms/LoopVectorize/reduction-inloop-pred.ll
index c0364c6fc5032..2c5459d472328 100644
--- a/llvm/test/Transforms/LoopVectorize/reduction-inloop-pred.ll
+++ b/llvm/test/Transforms/LoopVectorize/reduction-inloop-pred.ll
@@ -667,16 +667,14 @@ define float @reduction_conditional(ptr %A, ptr %B, ptr %C, float %S) {
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x float>, ptr [[TMP2]], align 4
; CHECK-NEXT: [[TMP3:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD]], [[WIDE_LOAD1]]
; CHECK-NEXT: [[TMP4:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD1]], splat (float 1.000000e+00)
-; CHECK-NEXT: [[TMP8:%.*]] = xor <4 x i1> [[TMP4]], splat (i1 true)
-; CHECK-NEXT: [[TMP6:%.*]] = fcmp ule <4 x float> [[WIDE_LOAD]], splat (float 2.000000e+00)
+; CHECK-NEXT: [[TMP6:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD]], splat (float 2.000000e+00)
; CHECK-NEXT: [[TMP7:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD1]]
; CHECK-NEXT: [[TMP5:%.*]] = and <4 x i1> [[TMP3]], [[TMP4]]
; CHECK-NEXT: [[TMP9:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD]]
-; CHECK-NEXT: [[TMP10:%.*]] = and <4 x i1> [[TMP6]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = or <4 x i1> [[TMP4]], [[TMP6]]
; CHECK-NEXT: [[TMP11:%.*]] = and <4 x i1> [[TMP10]], [[TMP3]]
-; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP11]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP7]]
-; CHECK-NEXT: [[PREDPHI2:%.*]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
-; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP3]], <4 x float> [[PREDPHI2]], <4 x float> [[VEC_PHI]]
+; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP11]], <4 x float> [[TMP7]], <4 x float> [[VEC_PHI]]
+; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 128
; CHECK-NEXT: br i1 [[TMP12]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
diff --git a/llvm/test/Transforms/LoopVectorize/reduction-inloop.ll b/llvm/test/Transforms/LoopVectorize/reduction-inloop.ll
index d9d30f8f3e0a5..7f932a771d55a 100644
--- a/llvm/test/Transforms/LoopVectorize/reduction-inloop.ll
+++ b/llvm/test/Transforms/LoopVectorize/reduction-inloop.ll
@@ -1097,10 +1097,11 @@ define float @reduction_conditional(ptr %A, ptr %B, ptr %C, float %S) {
; CHECK-NEXT: [[TMP7:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD1]]
; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP3]], <4 x i1> [[TMP4]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP9:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD]]
+; CHECK-NEXT: [[TMP14:%.*]] = xor <4 x i1> [[TMP3]], splat (i1 true)
; CHECK-NEXT: [[TMP11:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP6]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP11]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP7]]
-; CHECK-NEXT: [[PREDPHI2:%.*]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
-; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP3]], <4 x float> [[PREDPHI2]], <4 x float> [[VEC_PHI]]
+; CHECK-NEXT: [[TMP15:%.*]] = or <4 x i1> [[TMP11]], [[TMP14]]
+; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP15]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP7]]
+; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 128
; CHECK-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
@@ -1145,14 +1146,16 @@ define float @reduction_conditional(ptr %A, ptr %B, ptr %C, float %S) {
; CHECK-INTERLEAVED-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP6]], <4 x i1> [[TMP8]], <4 x i1> zeroinitializer
; CHECK-INTERLEAVED-NEXT: [[TMP17:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD]]
; CHECK-INTERLEAVED-NEXT: [[TMP18:%.*]] = fadd fast <4 x float> [[VEC_PHI1]], [[WIDE_LOAD2]]
+; CHECK-INTERLEAVED-NEXT: [[TMP27:%.*]] = xor <4 x i1> [[TMP5]], splat (i1 true)
+; CHECK-INTERLEAVED-NEXT: [[TMP28:%.*]] = xor <4 x i1> [[TMP6]], splat (i1 true)
; CHECK-INTERLEAVED-NEXT: [[TMP20:%.*]] = select <4 x i1> [[TMP19]], <4 x i1> [[TMP11]], <4 x i1> zeroinitializer
; CHECK-INTERLEAVED-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP21]], <4 x i1> [[TMP12]], <4 x i1> zeroinitializer
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP20]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP13]]
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI5:%.*]] = select <4 x i1> [[TMP15]], <4 x float> [[TMP17]], <4 x float> [[PREDPHI]]
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI6]] = select <4 x i1> [[TMP5]], <4 x float> [[PREDPHI5]], <4 x float> [[VEC_PHI]]
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI7:%.*]] = select <4 x i1> [[TMP22]], <4 x float> [[VEC_PHI1]], <4 x float> [[TMP14]]
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI8:%.*]] = select <4 x i1> [[TMP16]], <4 x float> [[TMP18]], <4 x float> [[PREDPHI7]]
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI9]] = select <4 x i1> [[TMP6]], <4 x float> [[PREDPHI8]], <4 x float> [[VEC_PHI1]]
+; CHECK-INTERLEAVED-NEXT: [[TMP25:%.*]] = or <4 x i1> [[TMP20]], [[TMP27]]
+; CHECK-INTERLEAVED-NEXT: [[TMP26:%.*]] = or <4 x i1> [[TMP22]], [[TMP28]]
+; CHECK-INTERLEAVED-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP25]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP13]]
+; CHECK-INTERLEAVED-NEXT: [[PREDPHI6]] = select <4 x i1> [[TMP15]], <4 x float> [[TMP17]], <4 x float> [[PREDPHI]]
+; CHECK-INTERLEAVED-NEXT: [[PREDPHI7:%.*]] = select <4 x i1> [[TMP26]], <4 x float> [[VEC_PHI1]], <4 x float> [[TMP14]]
+; CHECK-INTERLEAVED-NEXT: [[PREDPHI9]] = select <4 x i1> [[TMP16]], <4 x float> [[TMP18]], <4 x float> [[PREDPHI7]]
; CHECK-INTERLEAVED-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; CHECK-INTERLEAVED-NEXT: [[TMP23:%.*]] = icmp eq i64 [[INDEX_NEXT]], 128
; CHECK-INTERLEAVED-NEXT: br i1 [[TMP23]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
diff --git a/llvm/test/Transforms/LoopVectorize/reduction.ll b/llvm/test/Transforms/LoopVectorize/reduction.ll
index c683ef897715a..12ddde4698ead 100644
--- a/llvm/test/Transforms/LoopVectorize/reduction.ll
+++ b/llvm/test/Transforms/LoopVectorize/reduction.ll
@@ -761,16 +761,14 @@ define float @reduction_conditional(ptr %A, ptr %B, ptr %C, float %S) {
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x float>, ptr [[TMP2]], align 4
; CHECK-NEXT: [[TMP3:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD]], [[WIDE_LOAD1]]
; CHECK-NEXT: [[TMP4:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD1]], splat (float 1.000000e+00)
-; CHECK-NEXT: [[TMP8:%.*]] = xor <4 x i1> [[TMP4]], splat (i1 true)
-; CHECK-NEXT: [[TMP6:%.*]] = fcmp ule <4 x float> [[WIDE_LOAD]], splat (float 2.000000e+00)
+; CHECK-NEXT: [[TMP6:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD]], splat (float 2.000000e+00)
; CHECK-NEXT: [[TMP7:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD1]]
; CHECK-NEXT: [[TMP5:%.*]] = and <4 x i1> [[TMP3]], [[TMP4]]
; CHECK-NEXT: [[TMP9:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD]]
-; CHECK-NEXT: [[TMP10:%.*]] = and <4 x i1> [[TMP6]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = or <4 x i1> [[TMP4]], [[TMP6]]
; CHECK-NEXT: [[TMP11:%.*]] = and <4 x i1> [[TMP10]], [[TMP3]]
-; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP11]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP7]]
-; CHECK-NEXT: [[PREDPHI2:%.*]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
-; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP3]], <4 x float> [[PREDPHI2]], <4 x float> [[VEC_PHI]]
+; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP11]], <4 x float> [[TMP7]], <4 x float> [[VEC_PHI]]
+; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 128
; CHECK-NEXT: br i1 [[TMP12]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
>From aedc836811a681bba6c829239f4be570ad48cc44 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Fri, 5 Jun 2026 18:33:25 +0800
Subject: [PATCH 7/8] Remove unnecessary map_range
---
llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp | 5 ++---
1 file changed, 2 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 5f74b1cae0630..5111fe2b6e0ca 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -331,9 +331,8 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
VPValue *VPPredicator::createMaskDisjunction(ArrayRef<EdgeTy> Edges,
VPBasicBlock *VPBB) {
- auto Dsts = map_range(Edges, [](auto E) { return E.second; });
- const VPBasicBlock *PostDom = *Dsts.begin();
- for (const VPBasicBlock *VPBB : drop_begin(Dsts))
+ const VPBasicBlock *PostDom = Edges[0].second;
+ for (auto [_, VPBB] : drop_begin(Edges))
PostDom =
cast<VPBasicBlock>(VPPDT.findNearestCommonDominator(PostDom, VPBB));
assert(VPPDT.dominates(VPBB, PostDom) && "Edges don't postdominate VPBB");
>From 20f97bd6eaa021c8587dfab1bbc9b651c3621c5f Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Thu, 4 Jun 2026 13:10:07 +0800
Subject: [PATCH 8/8] Remove masked cond
---
llvm/lib/Transforms/Vectorize/VPlan.h | 3 -
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 6 -
.../Transforms/Vectorize/VPlanTransforms.cpp | 150 ++++++++++++++----
llvm/lib/Transforms/Vectorize/VPlanUtils.cpp | 10 +-
.../Transforms/Vectorize/VPlanVerifier.cpp | 18 ---
.../predicated-early-exits-interleave.ll | 18 +--
.../predicated-multiple-exits.ll | 42 +++--
.../LoopVectorize/single_early_exit.ll | 62 ++++++++
.../Vectorize/VPlanUncountableExitTest.cpp | 7 +-
9 files changed, 217 insertions(+), 99 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlan.h b/llvm/lib/Transforms/Vectorize/VPlan.h
index b740665fe70bf..8d6be4aa3e465 100644
--- a/llvm/lib/Transforms/Vectorize/VPlan.h
+++ b/llvm/lib/Transforms/Vectorize/VPlan.h
@@ -1380,9 +1380,6 @@ class LLVM_ABI_FOR_TEST VPInstruction : public VPRecipeWithIRFlags,
/// Returns true if the VPInstruction does not need masking.
bool alwaysUnmasked() const {
- if (Opcode == VPInstruction::MaskedCond)
- return false;
-
// For now only VPInstructions with underlying values use masks.
// TODO: provide masks to VPInstructions w/o underlying values.
if (!getUnderlyingValue())
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index c3867024c34dc..6c5b4c9f87b03 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -475,7 +475,6 @@ Type *llvm::computeScalarTypeForInstruction(unsigned Opcode,
return IntegerType::get(Ctx, 1);
case VPInstruction::LogicalAnd:
case VPInstruction::LogicalOr:
- case VPInstruction::MaskedCond:
assert((!Op0Ty || Op0Ty->isIntegerTy(1)) && "expected bool operand");
AssertOperandType(1, Op0Ty);
return IntegerType::get(Ctx, 1);
@@ -583,7 +582,6 @@ unsigned VPInstruction::getNumOperandsForOpcode() const {
case VPInstruction::ExtractLastLane:
case VPInstruction::ExtractLastPart:
case VPInstruction::ExtractPenultimateElement:
- case VPInstruction::MaskedCond:
case VPInstruction::Not:
case VPInstruction::ResumeForEpilogue:
case VPInstruction::Reverse:
@@ -1533,7 +1531,6 @@ bool VPInstruction::opcodeMayReadOrWriteFromMemory() const {
case VPInstruction::FirstOrderRecurrenceSplice:
case VPInstruction::LogicalAnd:
case VPInstruction::LogicalOr:
- case VPInstruction::MaskedCond:
case VPInstruction::Not:
case VPInstruction::PtrAdd:
case VPInstruction::WideIVStep:
@@ -1682,9 +1679,6 @@ void VPInstruction::printRecipe(raw_ostream &O, const Twine &Indent,
case VPInstruction::ExitingIVValue:
O << "exiting-iv-value";
break;
- case VPInstruction::MaskedCond:
- O << "masked-cond";
- break;
case VPInstruction::ExtractLane:
O << "extract-lane";
break;
diff --git a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
index 5df097628ba7f..dca77b3a3ea5d 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanTransforms.cpp
@@ -1669,11 +1669,6 @@ static void simplifyRecipe(VPSingleDefRecipe *Def) {
return;
}
- // Simplify MaskedCond with no block mask to its single operand.
- if (match(Def, m_VPInstruction<VPInstruction::MaskedCond>()) &&
- !cast<VPInstruction>(Def)->isMasked())
- return Def->replaceAllUsesWith(Def->getOperand(0));
-
// Look through ExtractLastLane.
if (match(Def, m_ExtractLastLane(m_VPValue(A)))) {
if (match(A, m_BuildVector())) {
@@ -2037,6 +2032,15 @@ static void simplifyBlends(VPlan &Plan) {
}
}
+ if (UniqueValues.size() == 2) {
+ for (unsigned I = 0; I != Blend->getNumIncomingValues(); ++I) {
+ if (match(Blend->getIncomingValue(I), m_False())) {
+ StartIndex = I;
+ break;
+ }
+ }
+ }
+
SmallVector<VPValue *, 4> OperandsWithMask;
OperandsWithMask.push_back(Blend->getIncomingValue(StartIndex));
@@ -4126,17 +4130,6 @@ void VPlanTransforms::convertToConcreteRecipes(VPlan &Plan) {
continue;
}
- // Lower MaskedCond with block mask to LogicalAnd.
- if (match(&R, m_VPInstruction<VPInstruction::MaskedCond>())) {
- auto *VPI = cast<VPInstruction>(&R);
- assert(VPI->isMasked() &&
- "Unmasked MaskedCond should be simplified earlier");
- VPI->replaceAllUsesWith(Builder.createNaryOp(
- VPInstruction::LogicalAnd, {VPI->getMask(), VPI->getOperand(0)}));
- VPI->eraseFromParent();
- continue;
- }
-
// Lower CanonicalIVIncrementForPart to plain Add.
if (match(
&R,
@@ -4206,6 +4199,7 @@ struct EarlyExitInfo {
VPBasicBlock *EarlyExitingVPBB;
VPIRBasicBlock *EarlyExitVPBB;
VPValue *CondToExit;
+ VPValue *ExitMask;
};
/// Update \p Plan to mask memory operations in the loop based on whether the
@@ -4364,10 +4358,64 @@ static bool handleUncountableExitsWithSideEffects(
return true;
}
+static VPValue *repairSSA(VPValue *Src, VPBasicBlock *SrcVPBB, VPValue *Other,
+ VPBasicBlock *VPBB, VPDominatorTree &VPDT,
+ DenseMap<VPBlockBase *, VPPhi *> &Phis) {
+
+ if (VPDT.dominates(SrcVPBB, VPBB))
+ return Src;
+ if (VPDT.dominates(VPBB, SrcVPBB))
+ return Other;
+ if (VPPhi *Phi = Phis.lookup(VPBB))
+ return Phi;
+
+ SmallVector<VPValue *> InVals;
+ for (auto *Pred : VPBB->predecessors())
+ InVals.push_back(
+ repairSSA(Src, SrcVPBB, Other, cast<VPBasicBlock>(Pred), VPDT, Phis));
+ if (all_equal(InVals))
+ return InVals[0];
+
+ VPPhi *Phi = VPBuilder(VPBB, VPBB->getFirstNonPhi()).createScalarPhi(InVals);
+ Phis[VPBB] = Phi;
+ return Phi;
+}
+
+/// Insert phi nodes to maintain SSA starting from \p VPBB, such that the
+/// resulting value is \p \Src on all paths that go through \p SrcVPBB, and \p
+/// Other otherwise.
+static VPValue *repairSSA(VPValue *Src, VPBasicBlock *SrcVPBB, VPValue *Other,
+ VPBasicBlock *VPBB, VPDominatorTree &VPDT) {
+ DenseMap<VPBlockBase *, VPPhi *> Phis;
+ return repairSSA(Src, SrcVPBB, Other, VPBB, VPDT, Phis);
+}
+
+/// Splits \p LatchVPBB so it only contains the IV increment recipes.
+static VPBasicBlock *splitLatchAtIVInc(VPBasicBlock *LatchVPBB) {
+ auto It = LatchVPBB->getTerminator()->getIterator();
+ while (!match(
+ It->getVPSingleValue(),
+ m_CombineOr(m_Add(m_Isa<VPWidenIntOrFpInductionRecipe>(), m_LiveIn()),
+ m_Sub(m_Isa<VPWidenIntOrFpInductionRecipe>(), m_LiveIn()),
+ m_VPInstruction<Instruction::GetElementPtr>(
+ m_Isa<VPWidenPointerInductionRecipe>(), m_LiveIn())))) {
+ if (It == LatchVPBB->begin())
+ return LatchVPBB;
+ It = std::prev(It);
+ }
+ LatchVPBB = LatchVPBB->splitAt(It);
+ LatchVPBB->setName("vector.latch");
+ return LatchVPBB;
+}
+
bool VPlanTransforms::handleUncountableEarlyExits(
VPlan &Plan, VPBasicBlock *HeaderVPBB, VPBasicBlock *LatchVPBB,
VPBasicBlock *MiddleVPBB, Loop *TheLoop, PredicatedScalarEvolution &PSE,
DominatorTree &DT, AssumptionCache *AC, UncountableExitStyle Style) {
+ // Split the latch at the IV increment so we can branch to it and predicate
+ // any recipes before the increment.
+ LatchVPBB = splitLatchAtIVInc(LatchVPBB);
+
VPDominatorTree VPDT(Plan);
VPBuilder LatchBuilder(LatchVPBB->getTerminator());
SmallVector<EarlyExitInfo> Exits;
@@ -4384,25 +4432,30 @@ bool VPlanTransforms::handleUncountableEarlyExits(
m_BranchOnCond(m_VPValue(CondOfEarlyExitingVPBB)));
assert(Matched && "Terminator must be BranchOnCond");
- // Insert the MaskedCond in the EarlyExitingVPBB so the predicator adds
- // the correct block mask.
VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
- auto *CondToEarlyExit = EarlyExitingBuilder.createNaryOp(
- VPInstruction::MaskedCond,
+ auto *CondToEarlyExit =
TrueSucc == ExitBlock
? CondOfEarlyExitingVPBB
- : EarlyExitingBuilder.createNot(CondOfEarlyExitingVPBB));
+ : EarlyExitingBuilder.createNot(CondOfEarlyExitingVPBB);
+
+ // Create the exit mask to predicate successors in other lanes.
+ // TODO: This mask doesn't get materialized yet because it's always folded
+ // away. Eventually we need to freeze it to account for the extra use.
+ VPValue *FirstExitLane =
+ EarlyExitingBuilder.createFirstActiveLane(CondToEarlyExit);
+ VPValue *ExitMask = EarlyExitingBuilder.createICmp(
+ CmpInst::ICMP_ULT,
+ EarlyExitingBuilder.createNaryOp(VPInstruction::StepVector, {},
+ FirstExitLane->getScalarType()),
+ FirstExitLane);
+
assert((isa<VPIRValue>(CondOfEarlyExitingVPBB) ||
!VPDT.properlyDominates(EarlyExitingVPBB, LatchVPBB) ||
VPDT.properlyDominates(
CondOfEarlyExitingVPBB->getDefiningRecipe()->getParent(),
LatchVPBB)) &&
"exit condition must dominate the latch");
- Exits.push_back({
- EarlyExitingVPBB,
- ExitBlock,
- CondToEarlyExit,
- });
+ Exits.push_back({EarlyExitingVPBB, ExitBlock, CondToEarlyExit, ExitMask});
}
}
@@ -4516,7 +4569,7 @@ bool VPlanTransforms::handleUncountableEarlyExits(
//
for (auto [Exit, VectorEarlyExitVPBB] :
zip_equal(Exits, VectorEarlyExitVPBBs)) {
- auto &[EarlyExitingVPBB, EarlyExitVPBB, _] = Exit;
+ auto &[EarlyExitingVPBB, EarlyExitVPBB, _, ExitMask] = Exit;
// Adjust the phi nodes in EarlyExitVPBB.
// 1. remove incoming values from EarlyExitingVPBB,
// 2. extract the incoming value at FirstActiveLane
@@ -4540,8 +4593,9 @@ bool VPlanTransforms::handleUncountableEarlyExits(
ExitIRI->addOperand(NewIncoming);
}
- EarlyExitingVPBB->getTerminator()->eraseFromParent();
+ EarlyExitingVPBB->getTerminator()->setOperand(0, ExitMask);
VPBlockUtils::disconnectBlocks(EarlyExitingVPBB, EarlyExitVPBB);
+ VPBlockUtils::connectBlocks(EarlyExitingVPBB, LatchVPBB);
VPBlockUtils::connectBlocks(VectorEarlyExitVPBB, EarlyExitVPBB);
}
@@ -4589,6 +4643,46 @@ bool VPlanTransforms::handleUncountableEarlyExits(
DispatchBuilder.setInsertPoint(CurrentBB);
}
+ // Repair any uses of CondToExit to preserve SSA.
+ VPDT.recalculate(Plan);
+ for (auto [I, Exit] : enumerate(Exits)) {
+ VPValue *Repaired = repairSSA(Exit.CondToExit, Exit.EarlyExitingVPBB,
+ Plan.getFalse(), LatchVPBB, VPDT);
+
+ // In the latch and dispatch blocks, CondToExit is only used by the logical
+ // or chain and the extracts. The incoming value is never read when a prior
+ // early exit is taken, so set it to poison on those edges.
+ VPValue *Poison =
+ Plan.getOrAddLiveIn(PoisonValue::get(Repaired->getScalarType()));
+ if (auto *Phi = dyn_cast<VPPhi>(Repaired))
+ for (unsigned J = 0; J < I; J++)
+ Phi->setIncomingValueForBlock(Exits[J].EarlyExitingVPBB, Poison);
+
+ Exit.CondToExit->replaceUsesWithIf(Repaired, [&](VPUser &U, unsigned I) {
+ auto &R = cast<VPRecipeBase>(U);
+ return VPDT.dominates(LatchVPBB, R.getParent()) &&
+ R.getVPSingleValue() != Repaired;
+ });
+ }
+
+ // Repair any live outs to preserve SSA.
+ SmallVector<VPBasicBlock *> LiveOutVPBBs = {MiddleVPBB};
+ append_range(LiveOutVPBBs, VectorEarlyExitVPBBs);
+ for (auto *LiveOutVPBB : LiveOutVPBBs)
+ for (VPRecipeBase &R : *LiveOutVPBB) {
+ VPValue *LiveOut;
+ if (!match(&R,
+ m_CombineOr(m_ExtractLastPart(m_VPValue(LiveOut)),
+ m_ExtractLane(m_VPValue(), m_VPValue(LiveOut)))))
+ continue;
+ VPValue *Poison =
+ Plan.getOrAddLiveIn(PoisonValue::get(LiveOut->getScalarType()));
+ VPValue *Repaired =
+ repairSSA(LiveOut, LiveOut->getDefiningRecipe()->getParent(), Poison,
+ LatchVPBB, VPDT);
+ R.replaceUsesOfWith(LiveOut, Repaired);
+ }
+
return true;
}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
index f7f2d4591a3ce..60734f15a3aec 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanUtils.cpp
@@ -600,7 +600,11 @@ vputils::getRecipesForUncountableExit(SmallVectorImpl<VPInstruction *> &Recipes,
// FIXME: Remove the single user restriction; it's here because we're
// starting with the simplest set of loops we can, and multiple
// users means needing to add PHI nodes in the transform.
- if (V->getNumUsers() > 1)
+ if (count_if(V->users(), [](VPUser *U) {
+ // Ignore VPInstruction::FirstActiveLanes inserted by
+ // handleUncountableEarlyExits
+ return !match(U, m_FirstActiveLane(m_VPValue()));
+ }) > 1)
return std::nullopt;
VPValue *Op1, *Op2;
@@ -621,10 +625,6 @@ vputils::getRecipesForUncountableExit(SmallVectorImpl<VPInstruction *> &Recipes,
Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
Recipes.push_back(cast<VPInstruction>(GepR));
GEPs.push_back(cast<VPInstruction>(GepR));
- } else if (match(V, m_VPInstruction<VPInstruction::MaskedCond>(
- m_VPValue(Op1)))) {
- Worklist.push_back(Op1);
- Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
} else
return std::nullopt;
}
diff --git a/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp b/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
index be7f84f3b5a00..a84e49e6d56c0 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanVerifier.cpp
@@ -337,11 +337,6 @@ bool VPlanVerifier::verifyVPBasicBlock(const VPBasicBlock *VPBB) {
return false;
}
- // MaskedCond may be used from blocks it don't dominate; the block will be
- // linearized and it will dominate its users after linearization.
- if (match(&R, m_VPInstruction<VPInstruction::MaskedCond>()))
- continue;
-
for (const VPUser *U : V->users()) {
auto *UI = cast<VPRecipeBase>(U);
if (isa<VPIRPhi>(UI) &&
@@ -386,19 +381,6 @@ bool VPlanVerifier::verifyVPBasicBlock(const VPBasicBlock *VPBB) {
continue;
}
- // Recipes in blocks with a MaskedCond may be used in exit blocks; the
- // block will be linearized and its recipes will dominate their users
- // after linearization.
- bool BlockHasMaskedCond = any_of(*VPBB, [](const VPRecipeBase &R) {
- return match(&R, m_VPInstruction<VPInstruction::MaskedCond>());
- });
- if (BlockHasMaskedCond &&
- any_of(VPBB->getPlan()->getExitBlocks(), [UI](VPIRBasicBlock *EB) {
- return is_contained(EB->getPredecessors(), UI->getParent());
- })) {
- continue;
- }
-
errs() << "Use before def!\n";
#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
VPSlotTracker Tracker(VPBB->getPlan());
diff --git a/llvm/test/Transforms/LoopVectorize/predicated-early-exits-interleave.ll b/llvm/test/Transforms/LoopVectorize/predicated-early-exits-interleave.ll
index df24b2838f6c4..d07ea380f55b8 100644
--- a/llvm/test/Transforms/LoopVectorize/predicated-early-exits-interleave.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicated-early-exits-interleave.ll
@@ -185,10 +185,8 @@ define i64 @three_early_exits() {
; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[GEP_A]], i64 4
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[GEP_A]], align 1
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
-; CHECK-NEXT: [[TMP2:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], splat (i8 -42)
-; CHECK-NEXT: [[TMP3:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD1]], splat (i8 -42)
-; CHECK-NEXT: [[TMP4:%.*]] = xor <4 x i1> [[TMP2]], splat (i1 true)
-; CHECK-NEXT: [[TMP5:%.*]] = xor <4 x i1> [[TMP3]], splat (i1 true)
+; CHECK-NEXT: [[TMP4:%.*]] = icmp sge <4 x i8> [[WIDE_LOAD]], splat (i8 -42)
+; CHECK-NEXT: [[TMP5:%.*]] = icmp sge <4 x i8> [[WIDE_LOAD1]], splat (i8 -42)
; CHECK-NEXT: [[TMP6:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], splat (i8 42)
; CHECK-NEXT: [[TMP7:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD1]], splat (i8 42)
; CHECK-NEXT: [[TMP8:%.*]] = xor <4 x i1> [[TMP6]], splat (i1 true)
@@ -201,8 +199,6 @@ define i64 @three_early_exits() {
; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP13]], align 1
; CHECK-NEXT: [[TMP14:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
; CHECK-NEXT: [[TMP15:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD1]], [[WIDE_LOAD3]]
-; CHECK-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[TMP17:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP15]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP18:%.*]] = select <4 x i1> [[TMP4]], <4 x i1> [[TMP6]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP19:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP20:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
@@ -211,16 +207,18 @@ define i64 @three_early_exits() {
; CHECK-NEXT: [[WIDE_LOAD5:%.*]] = load <4 x i8>, ptr [[TMP21]], align 1
; CHECK-NEXT: [[TMP22:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD4]]
; CHECK-NEXT: [[TMP23:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD1]], [[WIDE_LOAD5]]
-; CHECK-NEXT: [[TMP24:%.*]] = select <4 x i1> [[TMP18]], <4 x i1> [[TMP22]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP19]], <4 x i1> [[TMP23]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP26:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[TMP27:%.*]] = getelementptr inbounds i8, ptr [[TMP26]], i64 4
; CHECK-NEXT: [[WIDE_LOAD6:%.*]] = load <4 x i8>, ptr [[TMP26]], align 1
; CHECK-NEXT: [[WIDE_LOAD7:%.*]] = load <4 x i8>, ptr [[TMP27]], align 1
; CHECK-NEXT: [[TMP28:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD6]]
; CHECK-NEXT: [[TMP29:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD1]], [[WIDE_LOAD7]]
-; CHECK-NEXT: [[TMP30:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP28]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[TMP31:%.*]] = select <4 x i1> [[TMP3]], <4 x i1> [[TMP29]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP17:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP15]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP24:%.*]] = select <4 x i1> [[TMP18]], <4 x i1> [[TMP22]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP25:%.*]] = select <4 x i1> [[TMP19]], <4 x i1> [[TMP23]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP30:%.*]] = select <4 x i1> [[TMP4]], <4 x i1> zeroinitializer, <4 x i1> [[TMP28]]
+; CHECK-NEXT: [[TMP31:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> zeroinitializer, <4 x i1> [[TMP29]]
; CHECK-NEXT: [[TMP32:%.*]] = select <4 x i1> [[TMP16]], <4 x i1> splat (i1 true), <4 x i1> [[TMP24]]
; CHECK-NEXT: [[TMP33:%.*]] = select <4 x i1> [[TMP17]], <4 x i1> splat (i1 true), <4 x i1> [[TMP25]]
; CHECK-NEXT: [[TMP34:%.*]] = select <4 x i1> [[TMP32]], <4 x i1> splat (i1 true), <4 x i1> [[TMP30]]
diff --git a/llvm/test/Transforms/LoopVectorize/predicated-multiple-exits.ll b/llvm/test/Transforms/LoopVectorize/predicated-multiple-exits.ll
index f672f649915a4..ef282cd6af6f1 100644
--- a/llvm/test/Transforms/LoopVectorize/predicated-multiple-exits.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicated-multiple-exits.ll
@@ -17,15 +17,14 @@ define i64 @diamond_with_2_early_exits() {
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr @A, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = xor <4 x i1> [[TMP1]], splat (i1 true)
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP4]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[GEP_B]], align 1
; CHECK-NEXT: [[TMP7:%.*]] = zext <4 x i8> [[WIDE_LOAD2]] to <4 x i64>
; CHECK-NEXT: [[TMP8:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> zeroinitializer, <4 x i1> [[TMP4]]
; CHECK-NEXT: [[TMP9:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP8]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP10:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> splat (i1 true), <4 x i1> [[TMP9]]
; CHECK-NEXT: [[TMP11:%.*]] = freeze <4 x i1> [[TMP10]]
@@ -94,24 +93,23 @@ define i64 @three_early_exits() {
; CHECK-NEXT: [[IV:%.*]] = phi i64 [ 0, %[[LOOP_HEADER]] ], [ [[INDEX_NEXT:%.*]], %[[CHECK_B:.*]] ]
; CHECK-NEXT: [[GEP_A:%.*]] = getelementptr inbounds i8, ptr @A, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[GEP_A]], align 1
-; CHECK-NEXT: [[TMP1:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], splat (i8 -42)
-; CHECK-NEXT: [[TMP2:%.*]] = xor <4 x i1> [[TMP1]], splat (i1 true)
+; CHECK-NEXT: [[TMP2:%.*]] = icmp sge <4 x i8> [[WIDE_LOAD]], splat (i8 -42)
; CHECK-NEXT: [[TMP3:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], splat (i8 42)
; CHECK-NEXT: [[TMP4:%.*]] = xor <4 x i1> [[TMP3]], splat (i1 true)
; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP4]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP6:%.*]] = getelementptr inbounds i8, ptr @D, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[TMP6]], align 1
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP8:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP9:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP3]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP10]], align 1
; CHECK-NEXT: [[TMP11:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
-; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP11]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP13:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP13]], align 1
; CHECK-NEXT: [[TMP14:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD3]]
-; CHECK-NEXT: [[TMP15:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP8:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP11]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP15:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> zeroinitializer, <4 x i1> [[TMP14]]
; CHECK-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP8]], <4 x i1> splat (i1 true), <4 x i1> [[TMP12]]
; CHECK-NEXT: [[TMP17:%.*]] = select <4 x i1> [[TMP16]], <4 x i1> splat (i1 true), <4 x i1> [[TMP15]]
; CHECK-NEXT: [[TMP18:%.*]] = freeze <4 x i1> [[TMP17]]
@@ -193,11 +191,9 @@ define i64 @nested_diamond_inner_exits() {
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr @A, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = xor <4 x i1> [[TMP1]], splat (i1 true)
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr @D, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP4]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[GEP_B]], align 1
; CHECK-NEXT: [[TMP7:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD2]], zeroinitializer
@@ -206,9 +202,10 @@ define i64 @nested_diamond_inner_exits() {
; CHECK-NEXT: [[TMP10:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP10]], align 1
; CHECK-NEXT: [[TMP11:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD3]]
-; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP11]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP13:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP14:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> zeroinitializer, <4 x i1> [[TMP4]]
+; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP9]], <4 x i1> [[TMP11]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP15:%.*]] = select <4 x i1> [[TMP13]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> splat (i1 true), <4 x i1> [[TMP12]]
; CHECK-NEXT: [[TMP17:%.*]] = select <4 x i1> [[TMP16]], <4 x i1> splat (i1 true), <4 x i1> [[TMP15]]
@@ -297,14 +294,14 @@ define i64 @chain_of_3_exits() {
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[GEP_B]], align 1
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP4:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP3]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[GEP_C:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[GEP_C]], align 1
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
-; CHECK-NEXT: [[TMP7:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP6]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP8:%.*]] = getelementptr inbounds i8, ptr @D, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP8]], align 1
; CHECK-NEXT: [[TMP9:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD3]]
+; CHECK-NEXT: [[TMP4:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP3]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP7:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP6]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP10:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP9]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP11:%.*]] = select <4 x i1> [[TMP4]], <4 x i1> splat (i1 true), <4 x i1> [[TMP7]]
; CHECK-NEXT: [[TMP12:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> splat (i1 true), <4 x i1> [[TMP10]]
@@ -383,26 +380,24 @@ define i64 @four_exits_2x2_diamond() {
; CHECK-NEXT: [[TMP0:%.*]] = getelementptr inbounds i8, ptr @A, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP0]], align 1
; CHECK-NEXT: [[TMP1:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = xor <4 x i1> [[TMP1]], splat (i1 true)
; CHECK-NEXT: [[TMP3:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[TMP3]], align 1
; CHECK-NEXT: [[TMP4:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP2]], <4 x i1> [[TMP4]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[GEP_B]], align 1
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
-; CHECK-NEXT: [[TMP8:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP9:%.*]] = getelementptr inbounds i8, ptr @D, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD3:%.*]] = load <4 x i8>, ptr [[TMP9]], align 1
; CHECK-NEXT: [[TMP10:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD3]], zeroinitializer
-; CHECK-NEXT: [[TMP11:%.*]] = xor <4 x i1> [[TMP10]], splat (i1 true)
; CHECK-NEXT: [[TMP12:%.*]] = icmp ne <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD3]]
-; CHECK-NEXT: [[TMP13:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> [[TMP12]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP14:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD3]]
-; CHECK-NEXT: [[TMP15:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> zeroinitializer, <4 x i1> [[TMP4]]
+; CHECK-NEXT: [[TMP8:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[PREDPHI5:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> zeroinitializer, <4 x i1> [[TMP12]]
+; CHECK-NEXT: [[PREDPHI6:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP14]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP5]], <4 x i1> splat (i1 true), <4 x i1> [[TMP8]]
-; CHECK-NEXT: [[TMP17:%.*]] = select <4 x i1> [[TMP16]], <4 x i1> splat (i1 true), <4 x i1> [[TMP13]]
-; CHECK-NEXT: [[TMP18:%.*]] = select <4 x i1> [[TMP17]], <4 x i1> splat (i1 true), <4 x i1> [[TMP15]]
+; CHECK-NEXT: [[TMP11:%.*]] = select <4 x i1> [[TMP16]], <4 x i1> splat (i1 true), <4 x i1> [[PREDPHI5]]
+; CHECK-NEXT: [[TMP18:%.*]] = select <4 x i1> [[TMP11]], <4 x i1> splat (i1 true), <4 x i1> [[PREDPHI6]]
; CHECK-NEXT: [[TMP19:%.*]] = freeze <4 x i1> [[TMP18]]
; CHECK-NEXT: [[CMP1A:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP19]])
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[IV]], 4
@@ -420,7 +415,7 @@ define i64 @four_exits_2x2_diamond() {
; CHECK-NEXT: [[TMP23:%.*]] = extractelement <4 x i1> [[TMP8]], i64 [[FIRST_ACTIVE_LANE]]
; CHECK-NEXT: br i1 [[TMP23]], label %[[VECTOR_EARLY_EXIT_1:.*]], label %[[LOOP_LATCH:.*]]
; CHECK: [[LOOP_LATCH]]:
-; CHECK-NEXT: [[TMP24:%.*]] = extractelement <4 x i1> [[TMP13]], i64 [[FIRST_ACTIVE_LANE]]
+; CHECK-NEXT: [[TMP24:%.*]] = extractelement <4 x i1> [[PREDPHI5]], i64 [[FIRST_ACTIVE_LANE]]
; CHECK-NEXT: br i1 [[TMP24]], label %[[VECTOR_EARLY_EXIT_2:.*]], label %[[VECTOR_EARLY_EXIT_3:.*]]
; CHECK: [[VECTOR_EARLY_EXIT_3]]:
; CHECK-NEXT: br label %[[LOOP_END1]]
@@ -577,10 +572,9 @@ define i64 @diamond_exits_overlapping_conditions() {
; CHECK-NEXT: [[GEP_C:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[GEP_C]], align 1
; CHECK-NEXT: [[TMP3:%.*]] = icmp slt <4 x i8> [[WIDE_LOAD]], zeroinitializer
-; CHECK-NEXT: [[TMP4:%.*]] = xor <4 x i1> [[TMP3]], splat (i1 true)
; CHECK-NEXT: [[TMP5:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
-; CHECK-NEXT: [[TMP6:%.*]] = select <4 x i1> [[TMP4]], <4 x i1> [[TMP5]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP7:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
+; CHECK-NEXT: [[TMP6:%.*]] = select <4 x i1> [[TMP3]], <4 x i1> zeroinitializer, <4 x i1> [[TMP5]]
; CHECK-NEXT: [[TMP8:%.*]] = select <4 x i1> [[TMP3]], <4 x i1> [[TMP7]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP9:%.*]] = select <4 x i1> [[TMP6]], <4 x i1> splat (i1 true), <4 x i1> [[TMP8]]
; CHECK-NEXT: [[TMP10:%.*]] = freeze <4 x i1> [[TMP9]]
@@ -656,10 +650,10 @@ define i64 @exit_from_merge_of_exit_fallthrough_and_bypass() {
; CHECK-NEXT: [[GEP_B:%.*]] = getelementptr inbounds i8, ptr @B, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x i8>, ptr [[GEP_B]], align 1
; CHECK-NEXT: [[TMP3:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD1]]
-; CHECK-NEXT: [[TMP4:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP3]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[GEP_C:%.*]] = getelementptr inbounds i8, ptr @C, i64 [[IV]]
; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[GEP_C]], align 1
; CHECK-NEXT: [[TMP6:%.*]] = icmp eq <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP4:%.*]] = select <4 x i1> [[TMP1]], <4 x i1> [[TMP3]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP7:%.*]] = select <4 x i1> [[TMP4]], <4 x i1> splat (i1 true), <4 x i1> [[TMP6]]
; CHECK-NEXT: [[TMP8:%.*]] = freeze <4 x i1> [[TMP7]]
; CHECK-NEXT: [[CMP_C:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP8]])
diff --git a/llvm/test/Transforms/LoopVectorize/single_early_exit.ll b/llvm/test/Transforms/LoopVectorize/single_early_exit.ll
index 3848ac68a07c8..e90d95e451bfa 100644
--- a/llvm/test/Transforms/LoopVectorize/single_early_exit.ll
+++ b/llvm/test/Transforms/LoopVectorize/single_early_exit.ll
@@ -620,6 +620,67 @@ exit:
%res = phi i64 [ -1, %entry ], [ -2, %then ], [ 0, %loop.latch ], [ %iv, %loop.header ]
ret i64 %res
}
+
+define i64 @same_exit_block_phi_of_consts_iv_next_and() {
+; CHECK-LABEL: define i64 @same_exit_block_phi_of_consts_iv_next_and() {
+; CHECK-NEXT: entry:
+; CHECK-NEXT: [[P1:%.*]] = alloca [1024 x i8], align 1
+; CHECK-NEXT: [[P2:%.*]] = alloca [1024 x i8], align 1
+; CHECK-NEXT: call void @init_mem(ptr [[P1]], i64 1024)
+; CHECK-NEXT: call void @init_mem(ptr [[P2]], i64 1024)
+; CHECK-NEXT: br label [[VECTOR_PH:%.*]]
+; CHECK: vector.ph:
+; CHECK-NEXT: br label [[VECTOR_BODY:%.*]]
+; CHECK: vector.body:
+; CHECK-NEXT: [[INDEX1:%.*]] = phi i64 [ 0, [[VECTOR_PH]] ], [ [[INDEX_NEXT3:%.*]], [[VECTOR_BODY_INTERIM:%.*]] ]
+; CHECK-NEXT: [[TMP0:%.*]] = add i64 3, [[INDEX1]]
+; CHECK-NEXT: [[TMP1:%.*]] = getelementptr inbounds i8, ptr [[P1]], i64 [[TMP0]]
+; CHECK-NEXT: [[WIDE_LOAD:%.*]] = load <4 x i8>, ptr [[TMP1]], align 1
+; CHECK-NEXT: [[TMP2:%.*]] = getelementptr inbounds i8, ptr [[P2]], i64 [[TMP0]]
+; CHECK-NEXT: [[WIDE_LOAD2:%.*]] = load <4 x i8>, ptr [[TMP2]], align 1
+; CHECK-NEXT: [[TMP3:%.*]] = icmp ne <4 x i8> [[WIDE_LOAD]], [[WIDE_LOAD2]]
+; CHECK-NEXT: [[TMP4:%.*]] = freeze <4 x i1> [[TMP3]]
+; CHECK-NEXT: [[TMP5:%.*]] = call i1 @llvm.vector.reduce.or.v4i1(<4 x i1> [[TMP4]])
+; CHECK-NEXT: [[INDEX_NEXT3]] = add nuw i64 [[INDEX1]], 4
+; CHECK-NEXT: [[TMP6:%.*]] = icmp eq i64 [[INDEX_NEXT3]], 64
+; CHECK-NEXT: br i1 [[TMP5]], label [[VECTOR_EARLY_EXIT:%.*]], label [[VECTOR_BODY_INTERIM]]
+; CHECK: vector.body.interim:
+; CHECK-NEXT: br i1 [[TMP6]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP13:![0-9]+]]
+; CHECK: middle.block:
+; CHECK-NEXT: br label [[LOOP_END:%.*]]
+; CHECK: vector.early.exit:
+; CHECK-NEXT: br label [[LOOP_END]]
+; CHECK: loop.end:
+; CHECK-NEXT: [[RETVAL:%.*]] = phi i64 [ 0, [[VECTOR_EARLY_EXIT]] ], [ 1, [[MIDDLE_BLOCK]] ]
+; CHECK-NEXT: ret i64 [[RETVAL]]
+;
+entry:
+ %p1 = alloca [1024 x i8]
+ %p2 = alloca [1024 x i8]
+ call void @init_mem(ptr %p1, i64 1024)
+ call void @init_mem(ptr %p2, i64 1024)
+ br label %loop
+
+loop:
+ %index = phi i64 [ %index.next, %loop.inc ], [ 3, %entry ]
+ %arrayidx = getelementptr inbounds i8, ptr %p1, i64 %index
+ %ld1 = load i8, ptr %arrayidx, align 1
+ %arrayidx1 = getelementptr inbounds i8, ptr %p2, i64 %index
+ %ld2 = load i8, ptr %arrayidx1, align 1
+ %cmp3 = icmp eq i8 %ld1, %ld2
+ br i1 %cmp3, label %loop.inc, label %loop.end
+
+loop.inc:
+ %index.next = add i64 %index, 1
+ %index.next.and = and i64 %index.next, 4294967295
+ %exitcond = icmp ne i64 %index.next.and, 67
+ br i1 %exitcond, label %loop, label %loop.end
+
+loop.end:
+ %retval = phi i64 [ 0, %loop ], [ 1, %loop.inc ]
+ ret i64 %retval
+}
+
;.
; CHECK: [[LOOP0]] = distinct !{[[LOOP0]], [[META1:![0-9]+]], [[META2:![0-9]+]]}
; CHECK: [[META1]] = !{!"llvm.loop.isvectorized", i32 1}
@@ -634,4 +695,5 @@ exit:
; CHECK: [[LOOP10]] = distinct !{[[LOOP10]], [[META2]], [[META1]]}
; CHECK: [[LOOP11]] = distinct !{[[LOOP11]], [[META1]], [[META2]]}
; CHECK: [[LOOP12]] = distinct !{[[LOOP12]], [[META2]], [[META1]]}
+; CHECK: [[LOOP13]] = distinct !{[[LOOP13]], [[META1]], [[META2]]}
;.
diff --git a/llvm/unittests/Transforms/Vectorize/VPlanUncountableExitTest.cpp b/llvm/unittests/Transforms/Vectorize/VPlanUncountableExitTest.cpp
index 3e10d5f17699d..a5af7e46f44f5 100644
--- a/llvm/unittests/Transforms/Vectorize/VPlanUncountableExitTest.cpp
+++ b/llvm/unittests/Transforms/Vectorize/VPlanUncountableExitTest.cpp
@@ -53,13 +53,10 @@ static void combineExitConditions(VPlan &Plan) {
VPBuilder EarlyExitBuilder(EarlyExitingVPBB->getTerminator());
if (EarlyExitingVPBB->getSuccessors()[0] != EarlyExitVPBB)
Cond = EarlyExitBuilder.createNot(Cond);
- auto *MaskedCond =
- EarlyExitBuilder.createNaryOp(VPInstruction::MaskedCond, {Cond});
// Combine the early exit with the latch exit on the latch terminator.
VPBuilder Builder(LatchVPBB->getTerminator());
- auto *IsAnyExitTaken =
- Builder.createNaryOp(VPInstruction::AnyOf, {MaskedCond});
+ auto *IsAnyExitTaken = Builder.createNaryOp(VPInstruction::AnyOf, {Cond});
auto *LatchBranch = cast<VPInstruction>(LatchVPBB->getTerminator());
assert(LatchBranch->getOpcode() == VPInstruction::BranchOnCond &&
"Unexpected terminator");
@@ -121,7 +118,7 @@ TEST_F(VPUncountableExitTest, FindUncountableExitRecipes) {
vputils::getRecipesForUncountableExit(Recipes, GEPs, LatchVPBB);
ASSERT_TRUE(UncountableCondition.has_value());
ASSERT_EQ(GEPs.size(), 1ull);
- ASSERT_EQ(Recipes.size(), 4ull);
+ ASSERT_EQ(Recipes.size(), 3ull);
}
TEST_F(VPUncountableExitTest, NoUncountableExit) {
More information about the llvm-commits
mailing list