[llvm] [VPlan] Peek through nested phi incoming values in computeBlendMasks (PR #203164)
Luke Lau via llvm-commits
llvm-commits at lists.llvm.org
Wed Jun 10 21:00:00 PDT 2026
https://github.com/lukel97 created https://github.com/llvm/llvm-project/pull/203164
Stacked on #201783
If we have a phi with an incoming value that is also a phi, and that inner phi postdominates the original phi, then we can "unfold" that inner phi and look through its incoming values.
This is needed to fold away the blend masks entirely when the SSA is repaired in #201784.
The two other changes in this PR:
- When a nested phi is unfolded, we may end with an unconditional incoming value, i.e. all paths from the header lead to it. In this case there'll be no edge mask, so handle this in the `if (InValEdges.size() == 1)` check.
- Predicated reduction chains can be optimized with this, but this leads to a lot of test diffs and requires some transforms like `handleFindLastReductions` to be updated. For now, just ignore any phis that are a part of a reduction chain so we can focus on them in a later PR.
>From 3bdeab2f1b47f8368f57c511dddf902c8d9bd45b Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Tue, 2 Jun 2026 18:01:09 +0800
Subject: [PATCH 01/10] Precommit tests
---
.../LoopVectorize/VPlan/predicator.ll | 82 +++++++++++++++++++
.../Transforms/LoopVectorize/predicator.ll | 69 ++++++++++++++++
2 files changed, 151 insertions(+)
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index 6f33d05b044e6..a9514684b4a9f 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -750,3 +750,85 @@ loop.latch:
exit:
ret void
}
+
+define void @simplifiable_blend(i1 %c1, i1 %c2, i1 %c3, i32 %x, i32 %y, ptr %p) {
+; CHECK-LABEL: VPlan for loop in 'simplifiable_blend'
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK-NEXT: Successor(s): B
+; CHECK-EMPTY:
+; CHECK-NEXT: B:
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = not ir<%c1>
+; CHECK-NEXT: Successor(s): E
+; CHECK-EMPTY:
+; CHECK-NEXT: E:
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = not ir<%c3>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and vp<[[VP4]]>, vp<[[VP5]]>
+; CHECK-NEXT: Successor(s): F
+; CHECK-EMPTY:
+; CHECK-NEXT: F:
+; CHECK-NEXT: Successor(s): A
+; CHECK-EMPTY:
+; CHECK-NEXT: A:
+; CHECK-NEXT: Successor(s): D
+; CHECK-EMPTY:
+; CHECK-NEXT: D:
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = not ir<%c2>
+; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = logical-and ir<%c1>, vp<[[VP7]]>
+; CHECK-NEXT: Successor(s): C
+; CHECK-EMPTY:
+; CHECK-NEXT: C:
+; CHECK-NEXT: EMIT vp<[[VP9:%[0-9]+]]> = logical-and ir<%c1>, ir<%c2>
+; CHECK-NEXT: Successor(s): latch
+; CHECK-EMPTY:
+; CHECK-NEXT: latch:
+; CHECK-NEXT: BLEND ir<%phi> = ir<%y>/vp<[[VP4]]> ir<%x>/vp<[[VP8]]> ir<%x>/vp<[[VP9]]>
+; CHECK-NEXT: EMIT ir<%gep> = getelementptr ir<%p>, ir<%iv>
+; CHECK-NEXT: EMIT store ir<%phi>, ir<%gep>
+; CHECK-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; CHECK-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<128>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK-NEXT: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [0, %entry], [%iv.next, %latch]
+ br i1 %c1, label %A, label %B
+
+A:
+ br i1 %c2, label %C, label %D
+
+B:
+ br i1 %c3, label %F, label %E
+
+C:
+ br label %latch
+
+D:
+ br label %latch
+
+E:
+ br label %F
+
+F:
+ br label %latch
+
+latch:
+ %phi = phi i32 [ %x, %C ], [ %x, %D ], [ %y, %F ]
+ %gep = getelementptr i32, ptr %p, i32 %iv
+ store i32 %phi, ptr %gep
+ %iv.next = add i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, 128
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/predicator.ll b/llvm/test/Transforms/LoopVectorize/predicator.ll
index 9760de76fc07f..36e6fb778bc3e 100644
--- a/llvm/test/Transforms/LoopVectorize/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicator.ll
@@ -231,3 +231,72 @@ bb3:
exit:
ret void
}
+
+define void @simplifiable_blend(i1 %c1, i1 %c2, i1 %c3, i32 %x, i32 %y, ptr %p) {
+; CHECK-LABEL: define void @simplifiable_blend(
+; CHECK-SAME: i1 [[C1:%.*]], i1 [[C2:%.*]], i1 [[C3:%.*]], i32 [[X:%.*]], i32 [[Y:%.*]], ptr [[P:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[Y]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT1]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <4 x i1> poison, i1 [[C2]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT4:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT3]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT5:%.*]] = insertelement <4 x i1> poison, i1 [[C1]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT6:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT5]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP0:%.*]] = xor <4 x i1> [[BROADCAST_SPLAT4]], splat (i1 true)
+; CHECK-NEXT: [[TMP1:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[TMP0]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[TMP2:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[BROADCAST_SPLAT4]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP1]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[PREDPHI7:%.*]] = select <4 x i1> [[TMP2]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[PREDPHI]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i32, ptr [[P]], i32 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[PREDPHI7]], ptr [[TMP3]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i32 [[INDEX_NEXT]], 128
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP5:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [0, %entry], [%iv.next, %latch]
+ br i1 %c1, label %A, label %B
+
+A:
+ br i1 %c2, label %C, label %D
+
+B:
+ br i1 %c3, label %F, label %E
+
+C:
+ br label %latch
+
+D:
+ br label %latch
+
+E:
+ br label %F
+
+F:
+ br label %latch
+
+latch:
+ %phi = phi i32 [ %x, %C ], [ %x, %D ], [ %y, %F ]
+ %gep = getelementptr i32, ptr %p, i32 %iv
+ store i32 %phi, ptr %gep
+ %iv.next = add i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, 128
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
>From e3b9c3012126d42f1f32f654e0711ee43b5bc9bf Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Wed, 3 Jun 2026 23:55:02 +0800
Subject: [PATCH 02/10] Simplify blend masks
---
.../include/llvm/Analysis/DominanceFrontier.h | 1 +
.../Transforms/Vectorize/VPlanDominatorTree.h | 8 +
.../Transforms/Vectorize/VPlanPredicator.cpp | 147 +++++++++++++++++-
.../LoopVectorize/VPlan/predicator.ll | 12 +-
.../LoopVectorize/predicate-switch.ll | 9 +-
.../Transforms/LoopVectorize/predicator.ll | 10 +-
.../LoopVectorize/reduction-inloop-pred.ll | 10 +-
.../LoopVectorize/reduction-inloop.ll | 21 +--
.../Transforms/LoopVectorize/reduction.ll | 10 +-
9 files changed, 181 insertions(+), 47 deletions(-)
diff --git a/llvm/include/llvm/Analysis/DominanceFrontier.h b/llvm/include/llvm/Analysis/DominanceFrontier.h
index fd38891e901e3..4a8ab96cf71a7 100644
--- a/llvm/include/llvm/Analysis/DominanceFrontier.h
+++ b/llvm/include/llvm/Analysis/DominanceFrontier.h
@@ -78,6 +78,7 @@ class DominanceFrontierBase {
const_iterator end() const { return Frontiers.end(); }
iterator find(BlockT *B) { return Frontiers.find(B); }
const_iterator find(BlockT *B) const { return Frontiers.find(B); }
+ const_iterator find(const BlockT *B) const { return Frontiers.find(B); }
/// print - Convert to human readable form
///
diff --git a/llvm/lib/Transforms/Vectorize/VPlanDominatorTree.h b/llvm/lib/Transforms/Vectorize/VPlanDominatorTree.h
index 2864670f44913..1ad522880c709 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanDominatorTree.h
+++ b/llvm/lib/Transforms/Vectorize/VPlanDominatorTree.h
@@ -18,6 +18,8 @@
#include "VPlan.h"
#include "VPlanCFG.h"
#include "llvm/ADT/GraphTraits.h"
+#include "llvm/Analysis/DominanceFrontier.h"
+#include "llvm/Analysis/DominanceFrontierImpl.h"
#include "llvm/IR/Dominators.h"
#include "llvm/Support/GenericDomTree.h"
#include "llvm/Support/GenericDomTreeConstruction.h"
@@ -67,5 +69,11 @@ template <>
struct GraphTraits<const VPDomTreeNode *>
: public DomTreeGraphTraitsBase<const VPDomTreeNode,
VPDomTreeNode::const_iterator> {};
+
+class VPPostDominanceFrontier
+ : public DominanceFrontierBase<VPBlockBase, true> {
+public:
+ explicit VPPostDominanceFrontier(const DomTreeT &VPDT) { analyze(VPDT); }
+};
} // namespace llvm
#endif // LLVM_TRANSFORMS_VECTORIZE_VPLANDOMINATORTREE_H
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 655ac58e24426..2a947df080a09 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -34,6 +34,9 @@ class VPPredicator {
/// Post-dominator tree for the VPlan.
VPPostDominatorTree VPPDT;
+ /// Post-dominator frontier for the VPlan.
+ VPPostDominanceFrontier VPPDF;
+
/// When we if-convert we need to create edge masks. We have to cache values
/// so that we don't end up with exponential recursion/IR.
using EdgeMaskCacheTy =
@@ -78,8 +81,17 @@ class VPPredicator {
return VPBB->getFirstNonPhi();
}
+ using EdgeTy = std::pair<const VPBasicBlock *, const VPBasicBlock *>;
+
+ /// Compute the "furthest up" set of edges for each incoming value of \Phi.
+ MapVector<EdgeTy, VPValue *> computeBlendEdges(VPPhi *Phi);
+
+ /// Given a set of \p Edges that lead to \p VPBB, return the OR of all edges
+ /// or an equivalent block in-mask.
+ VPValue *createMaskDisjunction(ArrayRef<EdgeTy> Edges, VPBasicBlock *VPBB);
+
public:
- VPPredicator(VPlan &Plan) : VPDT(Plan), VPPDT(Plan) {}
+ VPPredicator(VPlan &Plan) : VPDT(Plan), VPPDT(Plan), VPPDF(VPPDT) {}
/// Returns the *entry* mask for \p VPBB.
VPValue *getBlockInMask(const VPBasicBlock *VPBB) const {
@@ -233,6 +245,115 @@ void VPPredicator::createSwitchEdgeMasks(const VPInstruction *SI) {
setEdgeMask(Src, DefaultDst, DefaultMask);
}
+// Compute the "furthest up" set of edges for each incoming value of a phi.
+//
+// Start by keeping track of what edges lead to which value. Then see if any
+// node has the same value for all outgoing edges. If so then propagate that
+// value up to every node it postdominates.
+MapVector<VPPredicator::EdgeTy, VPValue *>
+VPPredicator::computeBlendEdges(VPPhi *Phi) {
+ MapVector<EdgeTy, VPValue *> Edges;
+
+ // Mark the given edge as providing the value \p V.
+ auto AddEdge = [&Edges](const VPBlockBase *From, const VPBlockBase *To,
+ VPValue *V) {
+ EdgeTy Edge = {cast<VPBasicBlock>(From), cast<VPBasicBlock>(To)};
+ assert((!Edges.contains(Edge) || Edges.lookup(Edge) == V) &&
+ "Clobbering an edge?");
+ Edges[Edge] = V;
+ };
+
+ for (auto [InVal, InVPBB] : Phi->incoming_values_and_blocks())
+ AddEdge(InVPBB, Phi->getParent(), InVal);
+
+ // The root phi must postdominate every incoming block. Also don't touch
+ // phis in a reduction chain since they need to be in a specific structure
+ // for handle*Reductions.
+ for (auto [InVal, InVPBB] : Phi->incoming_values_and_blocks())
+ if (!VPPDT.dominates(Phi->getParent(), InVPBB) ||
+ isa<VPReductionPHIRecipe>(InVal))
+ return Edges;
+
+ // Given a list of edges, check if they all have the same value and return it.
+ auto GetAllEqual = [&Edges](ArrayRef<EdgeTy> OutEdges) -> VPValue * {
+ VPValue *Common = nullptr;
+ for (EdgeTy E : OutEdges) {
+ VPValue *V = Edges.lookup(E);
+ if (!V)
+ return nullptr;
+ if (match(V, m_Poison()))
+ continue;
+ if (!Common)
+ Common = V;
+ else if (Common != V)
+ return nullptr;
+ }
+ return Common;
+ };
+
+ SetVector<const VPBlockBase *> Worklist(from_range, Phi->incoming_blocks());
+ while (!Worklist.empty()) {
+ auto *VPBB = cast<VPBasicBlock>(Worklist.pop_back_val());
+
+ // Check that all outgoing edges from VPBB have the same value.
+ SmallVector<EdgeTy> OutEdges;
+ for (const VPBlockBase *Succ : VPBB->getSuccessors())
+ OutEdges.emplace_back(VPBB, cast<VPBasicBlock>(Succ));
+ VPValue *Common = GetAllEqual(OutEdges);
+ if (!Common)
+ continue;
+
+ // They have the same value: we can move the edges up
+ for (EdgeTy Edge : OutEdges)
+ Edges.erase(Edge);
+
+ // Peek through phis that are postdominated by VPBB
+ if (auto *Phi = dyn_cast<VPPhi>(Common))
+ if (VPPDT.dominates(VPBB, Phi->getParent())) {
+ for (auto [InV, InVPBB] : Phi->incoming_values_and_blocks()) {
+ AddEdge(InVPBB, Phi->getParent(), InV);
+ Worklist.insert(InVPBB);
+ }
+ continue;
+ }
+
+ // Iterate up through the post dominance frontier
+ for (const VPBlockBase *Frontier : VPPDF.find(VPBB)->second) {
+ for (const VPBlockBase *FrontierSucc : Frontier->getSuccessors())
+ if (VPPDT.dominates(VPBB, FrontierSucc))
+ AddEdge(Frontier, FrontierSucc, Common);
+ Worklist.insert(cast<VPBasicBlock>(Frontier));
+ }
+ }
+
+ return Edges;
+}
+
+VPValue *VPPredicator::createMaskDisjunction(ArrayRef<EdgeTy> Edges,
+ VPBasicBlock *VPBB) {
+ auto Dsts = map_range(Edges, [](auto E) { return E.second; });
+ const VPBasicBlock *PostDom = *Dsts.begin();
+ for (const VPBasicBlock *VPBB : drop_begin(Dsts))
+ PostDom =
+ cast<VPBasicBlock>(VPPDT.findNearestCommonDominator(PostDom, VPBB));
+ assert(VPPDT.dominates(VPBB, PostDom) && "Edges don't postdominate VPBB");
+ if (PostDom != VPBB)
+ return getBlockInMask(PostDom);
+
+ VPValue *Mask = nullptr;
+ for (auto [Src, ConstDst] : Edges) {
+ auto *Dst = const_cast<VPBasicBlock *>(ConstDst);
+ VPValue *EdgeMask;
+ {
+ VPBuilder::InsertPointGuard Guard(Builder);
+ Builder.setInsertPoint(Dst, getMaskInsertPoint(Dst));
+ EdgeMask = createEdgeMask(Src, Dst);
+ }
+ Mask = Mask ? Builder.createOr(Mask, EdgeMask) : EdgeMask;
+ }
+ return Mask;
+}
+
void VPPredicator::convertPhisToBlends(VPBasicBlock *VPBB) {
Builder.setInsertPoint(VPBB, getMaskInsertPoint(VPBB));
@@ -256,10 +377,30 @@ void VPPredicator::convertPhisToBlends(VPBasicBlock *VPBB) {
continue;
}
+ MapVector<VPValue *, SmallVector<EdgeTy>> InValEdgesMap;
+ for (auto [Edge, Val] : computeBlendEdges(PhiR))
+ InValEdgesMap[Val].push_back(Edge);
+ auto InValEdges = InValEdgesMap.takeVector();
+
+ if (InValEdges.size() == 1) {
+ PhiR->replaceAllUsesWith(InValEdges[0].first);
+ PhiR->eraseFromParent();
+ continue;
+ }
+
+ // Sort the incoming value order to match PhiR as much as possible.
+ llvm::stable_sort(InValEdges, [&PhiR](auto &L, auto &R) {
+ auto InVs = PhiR->incoming_values();
+ return std::distance(InVs.begin(), find(InVs, L.first)) <
+ std::distance(InVs.begin(), find(InVs, R.first));
+ });
+
SmallVector<VPValue *, 2> OperandsWithMask;
- for (const auto &[InVPV, InVPBB] : PhiR->incoming_values_and_blocks()) {
+ for (const auto &[InVPV, Edges] : InValEdges) {
+ if (match(InVPV, m_Poison()))
+ continue;
OperandsWithMask.push_back(InVPV);
- OperandsWithMask.push_back(createEdgeMask(InVPBB, VPBB));
+ OperandsWithMask.push_back(createMaskDisjunction(Edges, VPBB));
}
PHINode *IRPhi = cast_or_null<PHINode>(PhiR->getUnderlyingValue());
auto *Blend =
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index a9514684b4a9f..a994367670242 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -309,7 +309,7 @@ define void @switch(ptr %a) {
; CHECK-NEXT: EMIT vp<[[VP13:%[0-9]+]]> = not vp<[[VP12]]>
; CHECK-NEXT: EMIT vp<[[VP14:%[0-9]+]]> = logical-and ir<%c0>, vp<[[VP13]]>
; CHECK-NEXT: EMIT vp<[[VP15:%[0-9]+]]> = or vp<[[VP5]]>, vp<[[VP11]]>
-; CHECK-NEXT: BLEND ir<%phi3> = ir<%add2>/vp<[[VP5]]> ir<%add1>/vp<[[VP11]]> ir<%add1>/vp<[[VP11]]>
+; CHECK-NEXT: BLEND ir<%phi3> = ir<%add2>/vp<[[VP5]]> ir<%add1>/vp<[[VP11]]>
; CHECK-NEXT: EMIT ir<%add3> = add ir<%phi3>, ir<3>, vp<[[VP15]]>
; CHECK-NEXT: Successor(s): bb4
; CHECK-EMPTY:
@@ -665,8 +665,6 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): merge.a
; CHECK-EMPTY:
; CHECK-NEXT: merge.a:
-; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = not ir<%c0>
-; CHECK-NEXT: BLEND ir<%blend.a> = ir<%v1>/ir<%c0> ir<%v1>/vp<[[VP5]]>
; CHECK-NEXT: EMIT ir<%d0> = icmp sgt ir<%iv>, ir<0>
; CHECK-NEXT: Successor(s): if.b
; CHECK-EMPTY:
@@ -675,16 +673,14 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): if.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: if.b.inner:
-; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
; CHECK-NEXT: Successor(s): merge.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: merge.b.inner:
; CHECK-NEXT: Successor(s): merge.b
; CHECK-EMPTY:
; CHECK-NEXT: merge.b:
-; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = not ir<%d0>
-; CHECK-NEXT: BLEND ir<%blend.b> = ir<%v2>/ir<%d0> ir<%v2>/vp<[[VP7]]>
-; CHECK-NEXT: EMIT ir<%sum> = add ir<%blend.a>, ir<%blend.b>
+; CHECK-NEXT: EMIT ir<%sum> = add ir<%v1>, ir<%v2>
; CHECK-NEXT: EMIT store ir<%sum>, ir<%gep>
; CHECK-NEXT: Successor(s): loop.latch
; CHECK-EMPTY:
@@ -785,7 +781,7 @@ define void @simplifiable_blend(i1 %c1, i1 %c2, i1 %c3, i32 %x, i32 %y, ptr %p)
; CHECK-NEXT: Successor(s): latch
; CHECK-EMPTY:
; CHECK-NEXT: latch:
-; CHECK-NEXT: BLEND ir<%phi> = ir<%y>/vp<[[VP4]]> ir<%x>/vp<[[VP8]]> ir<%x>/vp<[[VP9]]>
+; CHECK-NEXT: BLEND ir<%phi> = ir<%y>/vp<[[VP4]]> ir<%x>/ir<%c1>
; CHECK-NEXT: EMIT ir<%gep> = getelementptr ir<%p>, ir<%iv>
; CHECK-NEXT: EMIT store ir<%phi>, ir<%gep>
; CHECK-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
diff --git a/llvm/test/Transforms/LoopVectorize/predicate-switch.ll b/llvm/test/Transforms/LoopVectorize/predicate-switch.ll
index 96775e0ff082e..d94b1d606e738 100644
--- a/llvm/test/Transforms/LoopVectorize/predicate-switch.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicate-switch.ll
@@ -545,8 +545,7 @@ define void @switch_unconditional_duplicate_target(ptr %start, ptr %dest) {
; IC1-NEXT: [[TMP4:%.*]] = extractelement <2 x ptr> [[TMP0]], i64 0
; IC1-NEXT: [[WIDE_LOAD:%.*]] = load <2 x i32>, ptr [[TMP4]], align 4, !alias.scope [[META6:![0-9]+]]
; IC1-NEXT: [[TMP5:%.*]] = icmp ult <2 x i32> [[WIDE_LOAD]], splat (i32 10)
-; IC1-NEXT: [[PREDPHI:%.*]] = select <2 x i1> [[TMP5]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP0]]
-; IC1-NEXT: [[PREDPHI2:%.*]] = select <2 x i1> [[TMP5]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[PREDPHI]]
+; IC1-NEXT: [[PREDPHI2:%.*]] = select <2 x i1> [[TMP5]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP0]]
; IC1-NEXT: [[TMP1:%.*]] = extractelement <2 x ptr> [[PREDPHI2]], i64 0
; IC1-NEXT: [[TMP2:%.*]] = extractelement <2 x ptr> [[PREDPHI2]], i64 1
; IC1-NEXT: store i32 0, ptr [[TMP1]], align 4
@@ -605,12 +604,10 @@ define void @switch_unconditional_duplicate_target(ptr %start, ptr %dest) {
; IC2-NEXT: [[WIDE_LOAD2:%.*]] = load <2 x i32>, ptr [[TMP8]], align 4, !alias.scope [[META6]]
; IC2-NEXT: [[TMP9:%.*]] = icmp ult <2 x i32> [[WIDE_LOAD]], splat (i32 10)
; IC2-NEXT: [[TMP10:%.*]] = icmp ult <2 x i32> [[WIDE_LOAD2]], splat (i32 10)
-; IC2-NEXT: [[PREDPHI:%.*]] = select <2 x i1> [[TMP9]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP0]]
-; IC2-NEXT: [[PREDPHI2:%.*]] = select <2 x i1> [[TMP9]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[PREDPHI]]
+; IC2-NEXT: [[PREDPHI2:%.*]] = select <2 x i1> [[TMP9]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP0]]
; IC2-NEXT: [[TMP2:%.*]] = extractelement <2 x ptr> [[PREDPHI2]], i64 0
; IC2-NEXT: [[TMP3:%.*]] = extractelement <2 x ptr> [[PREDPHI2]], i64 1
-; IC2-NEXT: [[PREDPHI5:%.*]] = select <2 x i1> [[TMP10]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP1]]
-; IC2-NEXT: [[PREDPHI4:%.*]] = select <2 x i1> [[TMP10]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[PREDPHI5]]
+; IC2-NEXT: [[PREDPHI4:%.*]] = select <2 x i1> [[TMP10]], <2 x ptr> [[BROADCAST_SPLAT]], <2 x ptr> [[TMP1]]
; IC2-NEXT: [[TMP4:%.*]] = extractelement <2 x ptr> [[PREDPHI4]], i64 0
; IC2-NEXT: [[TMP5:%.*]] = extractelement <2 x ptr> [[PREDPHI4]], i64 1
; IC2-NEXT: store i32 0, ptr [[TMP2]], align 4
diff --git a/llvm/test/Transforms/LoopVectorize/predicator.ll b/llvm/test/Transforms/LoopVectorize/predicator.ll
index 36e6fb778bc3e..57414ae62c341 100644
--- a/llvm/test/Transforms/LoopVectorize/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicator.ll
@@ -242,15 +242,7 @@ define void @simplifiable_blend(i1 %c1, i1 %c2, i1 %c3, i32 %x, i32 %y, ptr %p)
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT1]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <4 x i1> poison, i1 [[C2]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT4:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT3]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT5:%.*]] = insertelement <4 x i1> poison, i1 [[C1]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT6:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT5]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP0:%.*]] = xor <4 x i1> [[BROADCAST_SPLAT4]], splat (i1 true)
-; CHECK-NEXT: [[TMP1:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[TMP0]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[TMP2:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[BROADCAST_SPLAT4]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP1]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[BROADCAST_SPLAT]]
-; CHECK-NEXT: [[PREDPHI7:%.*]] = select <4 x i1> [[TMP2]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[PREDPHI]]
+; CHECK-NEXT: [[PREDPHI7:%.*]] = select i1 [[C1]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
diff --git a/llvm/test/Transforms/LoopVectorize/reduction-inloop-pred.ll b/llvm/test/Transforms/LoopVectorize/reduction-inloop-pred.ll
index c0364c6fc5032..2c5459d472328 100644
--- a/llvm/test/Transforms/LoopVectorize/reduction-inloop-pred.ll
+++ b/llvm/test/Transforms/LoopVectorize/reduction-inloop-pred.ll
@@ -667,16 +667,14 @@ define float @reduction_conditional(ptr %A, ptr %B, ptr %C, float %S) {
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x float>, ptr [[TMP2]], align 4
; CHECK-NEXT: [[TMP3:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD]], [[WIDE_LOAD1]]
; CHECK-NEXT: [[TMP4:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD1]], splat (float 1.000000e+00)
-; CHECK-NEXT: [[TMP8:%.*]] = xor <4 x i1> [[TMP4]], splat (i1 true)
-; CHECK-NEXT: [[TMP6:%.*]] = fcmp ule <4 x float> [[WIDE_LOAD]], splat (float 2.000000e+00)
+; CHECK-NEXT: [[TMP6:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD]], splat (float 2.000000e+00)
; CHECK-NEXT: [[TMP7:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD1]]
; CHECK-NEXT: [[TMP5:%.*]] = and <4 x i1> [[TMP3]], [[TMP4]]
; CHECK-NEXT: [[TMP9:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD]]
-; CHECK-NEXT: [[TMP10:%.*]] = and <4 x i1> [[TMP6]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = or <4 x i1> [[TMP4]], [[TMP6]]
; CHECK-NEXT: [[TMP11:%.*]] = and <4 x i1> [[TMP10]], [[TMP3]]
-; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP11]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP7]]
-; CHECK-NEXT: [[PREDPHI2:%.*]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
-; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP3]], <4 x float> [[PREDPHI2]], <4 x float> [[VEC_PHI]]
+; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP11]], <4 x float> [[TMP7]], <4 x float> [[VEC_PHI]]
+; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 128
; CHECK-NEXT: br i1 [[TMP12]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
diff --git a/llvm/test/Transforms/LoopVectorize/reduction-inloop.ll b/llvm/test/Transforms/LoopVectorize/reduction-inloop.ll
index d9d30f8f3e0a5..7f932a771d55a 100644
--- a/llvm/test/Transforms/LoopVectorize/reduction-inloop.ll
+++ b/llvm/test/Transforms/LoopVectorize/reduction-inloop.ll
@@ -1097,10 +1097,11 @@ define float @reduction_conditional(ptr %A, ptr %B, ptr %C, float %S) {
; CHECK-NEXT: [[TMP7:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD1]]
; CHECK-NEXT: [[TMP5:%.*]] = select <4 x i1> [[TMP3]], <4 x i1> [[TMP4]], <4 x i1> zeroinitializer
; CHECK-NEXT: [[TMP9:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD]]
+; CHECK-NEXT: [[TMP14:%.*]] = xor <4 x i1> [[TMP3]], splat (i1 true)
; CHECK-NEXT: [[TMP11:%.*]] = select <4 x i1> [[TMP10]], <4 x i1> [[TMP6]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP11]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP7]]
-; CHECK-NEXT: [[PREDPHI2:%.*]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
-; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP3]], <4 x float> [[PREDPHI2]], <4 x float> [[VEC_PHI]]
+; CHECK-NEXT: [[TMP15:%.*]] = or <4 x i1> [[TMP11]], [[TMP14]]
+; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP15]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP7]]
+; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 128
; CHECK-NEXT: br i1 [[TMP12]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
@@ -1145,14 +1146,16 @@ define float @reduction_conditional(ptr %A, ptr %B, ptr %C, float %S) {
; CHECK-INTERLEAVED-NEXT: [[TMP16:%.*]] = select <4 x i1> [[TMP6]], <4 x i1> [[TMP8]], <4 x i1> zeroinitializer
; CHECK-INTERLEAVED-NEXT: [[TMP17:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD]]
; CHECK-INTERLEAVED-NEXT: [[TMP18:%.*]] = fadd fast <4 x float> [[VEC_PHI1]], [[WIDE_LOAD2]]
+; CHECK-INTERLEAVED-NEXT: [[TMP27:%.*]] = xor <4 x i1> [[TMP5]], splat (i1 true)
+; CHECK-INTERLEAVED-NEXT: [[TMP28:%.*]] = xor <4 x i1> [[TMP6]], splat (i1 true)
; CHECK-INTERLEAVED-NEXT: [[TMP20:%.*]] = select <4 x i1> [[TMP19]], <4 x i1> [[TMP11]], <4 x i1> zeroinitializer
; CHECK-INTERLEAVED-NEXT: [[TMP22:%.*]] = select <4 x i1> [[TMP21]], <4 x i1> [[TMP12]], <4 x i1> zeroinitializer
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP20]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP13]]
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI5:%.*]] = select <4 x i1> [[TMP15]], <4 x float> [[TMP17]], <4 x float> [[PREDPHI]]
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI6]] = select <4 x i1> [[TMP5]], <4 x float> [[PREDPHI5]], <4 x float> [[VEC_PHI]]
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI7:%.*]] = select <4 x i1> [[TMP22]], <4 x float> [[VEC_PHI1]], <4 x float> [[TMP14]]
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI8:%.*]] = select <4 x i1> [[TMP16]], <4 x float> [[TMP18]], <4 x float> [[PREDPHI7]]
-; CHECK-INTERLEAVED-NEXT: [[PREDPHI9]] = select <4 x i1> [[TMP6]], <4 x float> [[PREDPHI8]], <4 x float> [[VEC_PHI1]]
+; CHECK-INTERLEAVED-NEXT: [[TMP25:%.*]] = or <4 x i1> [[TMP20]], [[TMP27]]
+; CHECK-INTERLEAVED-NEXT: [[TMP26:%.*]] = or <4 x i1> [[TMP22]], [[TMP28]]
+; CHECK-INTERLEAVED-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP25]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP13]]
+; CHECK-INTERLEAVED-NEXT: [[PREDPHI6]] = select <4 x i1> [[TMP15]], <4 x float> [[TMP17]], <4 x float> [[PREDPHI]]
+; CHECK-INTERLEAVED-NEXT: [[PREDPHI7:%.*]] = select <4 x i1> [[TMP26]], <4 x float> [[VEC_PHI1]], <4 x float> [[TMP14]]
+; CHECK-INTERLEAVED-NEXT: [[PREDPHI9]] = select <4 x i1> [[TMP16]], <4 x float> [[TMP18]], <4 x float> [[PREDPHI7]]
; CHECK-INTERLEAVED-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 8
; CHECK-INTERLEAVED-NEXT: [[TMP23:%.*]] = icmp eq i64 [[INDEX_NEXT]], 128
; CHECK-INTERLEAVED-NEXT: br i1 [[TMP23]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP15:![0-9]+]]
diff --git a/llvm/test/Transforms/LoopVectorize/reduction.ll b/llvm/test/Transforms/LoopVectorize/reduction.ll
index c683ef897715a..12ddde4698ead 100644
--- a/llvm/test/Transforms/LoopVectorize/reduction.ll
+++ b/llvm/test/Transforms/LoopVectorize/reduction.ll
@@ -761,16 +761,14 @@ define float @reduction_conditional(ptr %A, ptr %B, ptr %C, float %S) {
; CHECK-NEXT: [[WIDE_LOAD1:%.*]] = load <4 x float>, ptr [[TMP2]], align 4
; CHECK-NEXT: [[TMP3:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD]], [[WIDE_LOAD1]]
; CHECK-NEXT: [[TMP4:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD1]], splat (float 1.000000e+00)
-; CHECK-NEXT: [[TMP8:%.*]] = xor <4 x i1> [[TMP4]], splat (i1 true)
-; CHECK-NEXT: [[TMP6:%.*]] = fcmp ule <4 x float> [[WIDE_LOAD]], splat (float 2.000000e+00)
+; CHECK-NEXT: [[TMP6:%.*]] = fcmp ogt <4 x float> [[WIDE_LOAD]], splat (float 2.000000e+00)
; CHECK-NEXT: [[TMP7:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD1]]
; CHECK-NEXT: [[TMP5:%.*]] = and <4 x i1> [[TMP3]], [[TMP4]]
; CHECK-NEXT: [[TMP9:%.*]] = fadd fast <4 x float> [[VEC_PHI]], [[WIDE_LOAD]]
-; CHECK-NEXT: [[TMP10:%.*]] = and <4 x i1> [[TMP6]], [[TMP8]]
+; CHECK-NEXT: [[TMP10:%.*]] = or <4 x i1> [[TMP4]], [[TMP6]]
; CHECK-NEXT: [[TMP11:%.*]] = and <4 x i1> [[TMP10]], [[TMP3]]
-; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP11]], <4 x float> [[VEC_PHI]], <4 x float> [[TMP7]]
-; CHECK-NEXT: [[PREDPHI2:%.*]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
-; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP3]], <4 x float> [[PREDPHI2]], <4 x float> [[VEC_PHI]]
+; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP11]], <4 x float> [[TMP7]], <4 x float> [[VEC_PHI]]
+; CHECK-NEXT: [[PREDPHI3]] = select <4 x i1> [[TMP5]], <4 x float> [[TMP9]], <4 x float> [[PREDPHI]]
; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i64 [[INDEX]], 4
; CHECK-NEXT: [[TMP12:%.*]] = icmp eq i64 [[INDEX_NEXT]], 128
; CHECK-NEXT: br i1 [[TMP12]], label [[MIDDLE_BLOCK:%.*]], label [[VECTOR_BODY]], !llvm.loop [[LOOP20:![0-9]+]]
>From 49eabde306a941e3daefc46d0f22e1c7c1583630 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Fri, 5 Jun 2026 18:33:25 +0800
Subject: [PATCH 03/10] Remove unnecessary map_range
---
llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp | 5 ++---
1 file changed, 2 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 2a947df080a09..711230f0d10a5 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -331,9 +331,8 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
VPValue *VPPredicator::createMaskDisjunction(ArrayRef<EdgeTy> Edges,
VPBasicBlock *VPBB) {
- auto Dsts = map_range(Edges, [](auto E) { return E.second; });
- const VPBasicBlock *PostDom = *Dsts.begin();
- for (const VPBasicBlock *VPBB : drop_begin(Dsts))
+ const VPBasicBlock *PostDom = Edges[0].second;
+ for (auto [_, VPBB] : drop_begin(Edges))
PostDom =
cast<VPBasicBlock>(VPPDT.findNearestCommonDominator(PostDom, VPBB));
assert(VPPDT.dominates(VPBB, PostDom) && "Edges don't postdominate VPBB");
>From 3724b66ef0f6aa2dd1c90a77a81808780309fadd Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Tue, 9 Jun 2026 18:54:40 +0800
Subject: [PATCH 04/10] Address review comments, add ASCII diagram and comments
---
.../Transforms/Vectorize/VPlanPredicator.cpp | 44 ++++++++++++++-----
1 file changed, 33 insertions(+), 11 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 711230f0d10a5..086605ee21e75 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -86,8 +86,8 @@ class VPPredicator {
/// Compute the "furthest up" set of edges for each incoming value of \Phi.
MapVector<EdgeTy, VPValue *> computeBlendEdges(VPPhi *Phi);
- /// Given a set of \p Edges that lead to \p VPBB, return the OR of all edges
- /// or an equivalent block in-mask.
+ /// Given a set of \p Edges that each can reach \p VPBB, return the OR of all
+ /// edges, or an equivalent block in-mask.
VPValue *createMaskDisjunction(ArrayRef<EdgeTy> Edges, VPBasicBlock *VPBB);
public:
@@ -245,11 +245,19 @@ void VPPredicator::createSwitchEdgeMasks(const VPInstruction *SI) {
setEdgeMask(Src, DefaultDst, DefaultMask);
}
-// Compute the "furthest up" set of edges for each incoming value of a phi.
-//
// Start by keeping track of what edges lead to which value. Then see if any
// node has the same value for all outgoing edges. If so then propagate that
-// value up to every node it postdominates.
+// value up to every node it postdominates. E.g:
+//
+// Entry Edges = {C->ɸ : %x, D->ɸ : %x, F->ɸ : %y}
+// / \ [C,D,F all outgoing edges equal: go up postdom frontier]
+// A B ~> {A->C : %x, A->D : %x, Entry->B : %y}
+// / \ |\ [A all outgoing edges equal: go up postdom frontier]
+// C D | E ~> {Entry->A : %x, Entry->B : %y}
+// \ \ |/
+// \ | F
+// \ | /
+// ɸ = phi [%x, C], [%x, D], [%y, F]
MapVector<VPPredicator::EdgeTy, VPValue *>
VPPredicator::computeBlendEdges(VPPhi *Phi) {
MapVector<EdgeTy, VPValue *> Edges;
@@ -266,9 +274,9 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
for (auto [InVal, InVPBB] : Phi->incoming_values_and_blocks())
AddEdge(InVPBB, Phi->getParent(), InVal);
- // The root phi must postdominate every incoming block. Also don't touch
- // phis in a reduction chain since they need to be in a specific structure
- // for handle*Reductions.
+ // Only handle phis that postdominate every incoming block. Also don't touch
+ // phis in a reduction chain since they need to be in a specific structure for
+ // handle*Reductions.
for (auto [InVal, InVPBB] : Phi->incoming_values_and_blocks())
if (!VPPDT.dominates(Phi->getParent(), InVPBB) ||
isa<VPReductionPHIRecipe>(InVal))
@@ -303,11 +311,11 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
if (!Common)
continue;
- // They have the same value: we can move the edges up
+ // They have the same value: we can move the edges up.
for (EdgeTy Edge : OutEdges)
Edges.erase(Edge);
- // Peek through phis that are postdominated by VPBB
+ // Peek through phis that are postdominated by VPBB.
if (auto *Phi = dyn_cast<VPPhi>(Common))
if (VPPDT.dominates(VPBB, Phi->getParent())) {
for (auto [InV, InVPBB] : Phi->incoming_values_and_blocks()) {
@@ -317,7 +325,7 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
continue;
}
- // Iterate up through the post dominance frontier
+ // Iterate up through the post dominance frontier.
for (const VPBlockBase *Frontier : VPPDF.find(VPBB)->second) {
for (const VPBlockBase *FrontierSucc : Frontier->getSuccessors())
if (VPPDT.dominates(VPBB, FrontierSucc))
@@ -331,6 +339,19 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
VPValue *VPPredicator::createMaskDisjunction(ArrayRef<EdgeTy> Edges,
VPBasicBlock *VPBB) {
+ // If the nearest common postdominator to all of Edges destinations isn't VPBB
+ // then we can use its block in-mask. E.g:
+ //
+ // A ... B
+ // \ \ /
+ // \ C
+ // \ /
+ // ... D ...
+ // \ | /
+ // VPBB
+ //
+ // If the edges are A->D and B->C, PostDom will be D. We can reuse Ds block
+ // in-mask.
const VPBasicBlock *PostDom = Edges[0].second;
for (auto [_, VPBB] : drop_begin(Edges))
PostDom =
@@ -339,6 +360,7 @@ VPValue *VPPredicator::createMaskDisjunction(ArrayRef<EdgeTy> Edges,
if (PostDom != VPBB)
return getBlockInMask(PostDom);
+ // Otherwise, compute the disjunction of edges.
VPValue *Mask = nullptr;
for (auto [Src, ConstDst] : Edges) {
auto *Dst = const_cast<VPBasicBlock *>(ConstDst);
>From 05ac6a4fec242720b0f2b60a29fbfb8fc66bcd78 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Tue, 9 Jun 2026 19:07:06 +0800
Subject: [PATCH 05/10] Remove peeking through phis
---
llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp | 10 ----------
llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll | 8 ++++++--
2 files changed, 6 insertions(+), 12 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 086605ee21e75..be3b2f39a9841 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -315,16 +315,6 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
for (EdgeTy Edge : OutEdges)
Edges.erase(Edge);
- // Peek through phis that are postdominated by VPBB.
- if (auto *Phi = dyn_cast<VPPhi>(Common))
- if (VPPDT.dominates(VPBB, Phi->getParent())) {
- for (auto [InV, InVPBB] : Phi->incoming_values_and_blocks()) {
- AddEdge(InVPBB, Phi->getParent(), InV);
- Worklist.insert(InVPBB);
- }
- continue;
- }
-
// Iterate up through the post dominance frontier.
for (const VPBlockBase *Frontier : VPPDF.find(VPBB)->second) {
for (const VPBlockBase *FrontierSucc : Frontier->getSuccessors())
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index a994367670242..74bd3ec92b78f 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -665,6 +665,8 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): merge.a
; CHECK-EMPTY:
; CHECK-NEXT: merge.a:
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = not ir<%c0>
+; CHECK-NEXT: BLEND ir<%blend.a> = ir<%v1>/ir<%c0> ir<%v1>/vp<[[VP5]]>
; CHECK-NEXT: EMIT ir<%d0> = icmp sgt ir<%iv>, ir<0>
; CHECK-NEXT: Successor(s): if.b
; CHECK-EMPTY:
@@ -673,14 +675,16 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): if.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: if.b.inner:
-; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
; CHECK-NEXT: Successor(s): merge.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: merge.b.inner:
; CHECK-NEXT: Successor(s): merge.b
; CHECK-EMPTY:
; CHECK-NEXT: merge.b:
-; CHECK-NEXT: EMIT ir<%sum> = add ir<%v1>, ir<%v2>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = not ir<%d0>
+; CHECK-NEXT: BLEND ir<%blend.b> = ir<%v2>/ir<%d0> ir<%v2>/vp<[[VP7]]>
+; CHECK-NEXT: EMIT ir<%sum> = add ir<%blend.a>, ir<%blend.b>
; CHECK-NEXT: EMIT store ir<%sum>, ir<%gep>
; CHECK-NEXT: Successor(s): loop.latch
; CHECK-EMPTY:
>From de9c116679a1c08986fb8e9c69d12df01c1718f5 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Wed, 10 Jun 2026 15:45:28 +0800
Subject: [PATCH 06/10] Remove reduction phi check
We don't need it for now
---
llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp | 9 +++------
1 file changed, 3 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index be3b2f39a9841..05c5d4a8e308e 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -274,12 +274,9 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
for (auto [InVal, InVPBB] : Phi->incoming_values_and_blocks())
AddEdge(InVPBB, Phi->getParent(), InVal);
- // Only handle phis that postdominate every incoming block. Also don't touch
- // phis in a reduction chain since they need to be in a specific structure for
- // handle*Reductions.
- for (auto [InVal, InVPBB] : Phi->incoming_values_and_blocks())
- if (!VPPDT.dominates(Phi->getParent(), InVPBB) ||
- isa<VPReductionPHIRecipe>(InVal))
+ // Only handle phis that postdominate every incoming block.
+ for (const VPBlockBase *InVPBB : Phi->incoming_blocks())
+ if (!VPPDT.dominates(Phi->getParent(), InVPBB))
return Edges;
// Given a list of edges, check if they all have the same value and return it.
>From d458213e69ac84cb4b209b1185579d58191236f9 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Wed, 10 Jun 2026 17:38:38 +0800
Subject: [PATCH 07/10] Fix assertion comment
---
llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 05c5d4a8e308e..4feb8d44b80f6 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -343,7 +343,7 @@ VPValue *VPPredicator::createMaskDisjunction(ArrayRef<EdgeTy> Edges,
for (auto [_, VPBB] : drop_begin(Edges))
PostDom =
cast<VPBasicBlock>(VPPDT.findNearestCommonDominator(PostDom, VPBB));
- assert(VPPDT.dominates(VPBB, PostDom) && "Edges don't postdominate VPBB");
+ assert(VPPDT.dominates(VPBB, PostDom) && "VPBB doesn't postdominate edges");
if (PostDom != VPBB)
return getBlockInMask(PostDom);
>From 0b553a5201ce649021ad88e9b0eb05aa3f12f3fc Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Wed, 10 Jun 2026 17:39:02 +0800
Subject: [PATCH 08/10] Remove poison coalescing for now until we have test
cases
---
llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp | 10 ----------
1 file changed, 10 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index 4feb8d44b80f6..e4891a6a86d8b 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -286,8 +286,6 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
VPValue *V = Edges.lookup(E);
if (!V)
return nullptr;
- if (match(V, m_Poison()))
- continue;
if (!Common)
Common = V;
else if (Common != V)
@@ -390,12 +388,6 @@ void VPPredicator::convertPhisToBlends(VPBasicBlock *VPBB) {
InValEdgesMap[Val].push_back(Edge);
auto InValEdges = InValEdgesMap.takeVector();
- if (InValEdges.size() == 1) {
- PhiR->replaceAllUsesWith(InValEdges[0].first);
- PhiR->eraseFromParent();
- continue;
- }
-
// Sort the incoming value order to match PhiR as much as possible.
llvm::stable_sort(InValEdges, [&PhiR](auto &L, auto &R) {
auto InVs = PhiR->incoming_values();
@@ -405,8 +397,6 @@ void VPPredicator::convertPhisToBlends(VPBasicBlock *VPBB) {
SmallVector<VPValue *, 2> OperandsWithMask;
for (const auto &[InVPV, Edges] : InValEdges) {
- if (match(InVPV, m_Poison()))
- continue;
OperandsWithMask.push_back(InVPV);
OperandsWithMask.push_back(createMaskDisjunction(Edges, VPBB));
}
>From bbe3155fb6a496be2b6f063579e2fb10fb9d3bb2 Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Tue, 9 Jun 2026 20:10:59 +0800
Subject: [PATCH 09/10] Precommit tests
---
.../LoopVectorize/VPlan/predicator.ll | 79 +++++++++++++++++++
.../Transforms/LoopVectorize/predicator.ll | 72 +++++++++++++++++
2 files changed, 151 insertions(+)
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index 74bd3ec92b78f..ba76b60903444 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -832,3 +832,82 @@ latch:
exit:
ret void
}
+
+; loop
+; / \
+; A \
+; / \ B
+; \ \ /
+; C D
+; \/
+; latch
+define void @look_thru_phi(i1 %c1, i1 %c2, i32 %x, i32 %y, ptr %p) {
+; CHECK-LABEL: VPlan for loop in 'look_thru_phi'
+; CHECK-NEXT: <x1> vector loop: {
+; CHECK-NEXT: vp<[[VP3:%[0-9]+]]> = CANONICAL-IV
+; CHECK-EMPTY:
+; CHECK-NEXT: vector.body:
+; CHECK-NEXT: ir<%iv> = WIDEN-INDUCTION ir<0>, ir<1>, vp<[[VP0:%[0-9]+]]>
+; CHECK-NEXT: Successor(s): B
+; CHECK-EMPTY:
+; CHECK-NEXT: B:
+; CHECK-NEXT: EMIT vp<[[VP4:%[0-9]+]]> = not ir<%c1>
+; CHECK-NEXT: Successor(s): A
+; CHECK-EMPTY:
+; CHECK-NEXT: A:
+; CHECK-NEXT: Successor(s): D
+; CHECK-EMPTY:
+; CHECK-NEXT: D:
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = not ir<%c2>
+; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and ir<%c1>, vp<[[VP5]]>
+; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = or vp<[[VP4]]>, vp<[[VP6]]>
+; CHECK-NEXT: BLEND ir<%phi1> = ir<%y>/vp<[[VP4]]> ir<%x>/vp<[[VP6]]>
+; CHECK-NEXT: Successor(s): C
+; CHECK-EMPTY:
+; CHECK-NEXT: C:
+; CHECK-NEXT: EMIT vp<[[VP8:%[0-9]+]]> = logical-and ir<%c1>, ir<%c2>
+; CHECK-NEXT: Successor(s): latch
+; CHECK-EMPTY:
+; CHECK-NEXT: latch:
+; CHECK-NEXT: BLEND ir<%phi> = ir<%phi1>/vp<[[VP7]]> ir<%x>/vp<[[VP8]]>
+; CHECK-NEXT: EMIT ir<%gep> = getelementptr ir<%p>, ir<%iv>
+; CHECK-NEXT: EMIT store ir<%phi>, ir<%gep>
+; CHECK-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
+; CHECK-NEXT: EMIT ir<%ec> = icmp eq ir<%iv.next>, ir<128>
+; CHECK-NEXT: EMIT vp<%index.next> = add nuw vp<[[VP3]]>, vp<[[VP1:%[0-9]+]]>
+; CHECK-NEXT: EMIT branch-on-count vp<%index.next>, vp<[[VP2:%[0-9]+]]>
+; CHECK-NEXT: No successors
+; CHECK-NEXT: }
+; CHECK-NEXT: Successor(s): middle.block
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [0, %entry], [%iv.next, %latch]
+ br i1 %c1, label %A, label %B
+
+A:
+ br i1 %c2, label %C, label %D
+
+B:
+ br label %D
+
+C:
+ br label %latch
+
+D:
+ %phi1 = phi i32 [%x, %A], [%y, %B]
+ br label %latch
+
+latch:
+ %phi = phi i32 [ %x, %C ], [ %phi1, %D ]
+ %gep = getelementptr i32, ptr %p, i32 %iv
+ store i32 %phi, ptr %gep
+ %iv.next = add i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, 128
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/Transforms/LoopVectorize/predicator.ll b/llvm/test/Transforms/LoopVectorize/predicator.ll
index 57414ae62c341..a03855edd8de7 100644
--- a/llvm/test/Transforms/LoopVectorize/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicator.ll
@@ -292,3 +292,75 @@ latch:
exit:
ret void
}
+
+; loop
+; / \
+; A \
+; / \ B
+; \ \ /
+; C D
+; \/
+; latch
+define void @look_thru_phi(i1 %c1, i1 %c2, i32 %x, i32 %y, ptr %p) {
+; CHECK-LABEL: define void @look_thru_phi(
+; CHECK-SAME: i1 [[C1:%.*]], i1 [[C2:%.*]], i32 [[X:%.*]], i32 [[Y:%.*]], ptr [[P:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: br label %[[VECTOR_PH:.*]]
+; CHECK: [[VECTOR_PH]]:
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <4 x i32> poison, i32 [[Y]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT1]], <4 x i32> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <4 x i1> poison, i1 [[C2]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT4:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT3]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[BROADCAST_SPLATINSERT5:%.*]] = insertelement <4 x i1> poison, i1 [[C1]], i64 0
+; CHECK-NEXT: [[BROADCAST_SPLAT6:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT5]], <4 x i1> poison, <4 x i32> zeroinitializer
+; CHECK-NEXT: [[TMP0:%.*]] = xor <4 x i1> [[BROADCAST_SPLAT4]], splat (i1 true)
+; CHECK-NEXT: [[TMP1:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[TMP0]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP1]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[BROADCAST_SPLAT]]
+; CHECK-NEXT: [[TMP2:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[BROADCAST_SPLAT4]], <4 x i1> zeroinitializer
+; CHECK-NEXT: [[PREDPHI7:%.*]] = select <4 x i1> [[TMP2]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[PREDPHI]]
+; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
+; CHECK: [[VECTOR_BODY]]:
+; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
+; CHECK-NEXT: [[TMP3:%.*]] = getelementptr i32, ptr [[P]], i32 [[INDEX]]
+; CHECK-NEXT: store <4 x i32> [[PREDPHI7]], ptr [[TMP3]], align 4
+; CHECK-NEXT: [[INDEX_NEXT]] = add nuw i32 [[INDEX]], 4
+; CHECK-NEXT: [[TMP4:%.*]] = icmp eq i32 [[INDEX_NEXT]], 128
+; CHECK-NEXT: br i1 [[TMP4]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
+; CHECK: [[MIDDLE_BLOCK]]:
+; CHECK-NEXT: br label %[[EXIT:.*]]
+; CHECK: [[EXIT]]:
+; CHECK-NEXT: ret void
+;
+entry:
+ br label %loop
+
+loop:
+ %iv = phi i32 [0, %entry], [%iv.next, %latch]
+ br i1 %c1, label %A, label %B
+
+A:
+ br i1 %c2, label %C, label %D
+
+B:
+ br label %D
+
+C:
+ br label %latch
+
+D:
+ %phi1 = phi i32 [%x, %A], [%y, %B]
+ br label %latch
+
+latch:
+ %phi = phi i32 [ %x, %C ], [ %phi1, %D ]
+ %gep = getelementptr i32, ptr %p, i32 %iv
+ store i32 %phi, ptr %gep
+ %iv.next = add i32 %iv, 1
+ %ec = icmp eq i32 %iv.next, 128
+ br i1 %ec, label %exit, label %loop
+
+exit:
+ ret void
+}
>From 0c700f0ec1b3300e46c7dc991a1fb76d450c2e7d Mon Sep 17 00:00:00 2001
From: Luke Lau <luke at igalia.com>
Date: Tue, 9 Jun 2026 20:12:38 +0800
Subject: [PATCH 10/10] Peek through inner phi's nested values
---
.../Transforms/Vectorize/VPlanPredicator.cpp | 21 +++++++++++++++++++
.../LoopVectorize/VPlan/predicator.ll | 10 +++------
.../Transforms/LoopVectorize/predicator.ll | 10 +--------
3 files changed, 25 insertions(+), 16 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
index e4891a6a86d8b..a1d2402b36e90 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanPredicator.cpp
@@ -279,6 +279,10 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
if (!VPPDT.dominates(Phi->getParent(), InVPBB))
return Edges;
+ // Don't optimize any reduction chains for now.
+ if (any_of(Phi->incoming_values(), IsaPred<VPReductionPHIRecipe>))
+ return Edges;
+
// Given a list of edges, check if they all have the same value and return it.
auto GetAllEqual = [&Edges](ArrayRef<EdgeTy> OutEdges) -> VPValue * {
VPValue *Common = nullptr;
@@ -310,6 +314,17 @@ VPPredicator::computeBlendEdges(VPPhi *Phi) {
for (EdgeTy Edge : OutEdges)
Edges.erase(Edge);
+ // If the value is a phi postdominated by VPBB, then look through the inner
+ // incoming values instead of propagating the phi.
+ if (auto *Phi = dyn_cast<VPPhi>(Common))
+ if (VPPDT.dominates(VPBB, Phi->getParent())) {
+ for (auto [InV, InVPBB] : Phi->incoming_values_and_blocks()) {
+ AddEdge(InVPBB, Phi->getParent(), InV);
+ Worklist.insert(InVPBB);
+ }
+ continue;
+ }
+
// Iterate up through the post dominance frontier.
for (const VPBlockBase *Frontier : VPPDF.find(VPBB)->second) {
for (const VPBlockBase *FrontierSucc : Frontier->getSuccessors())
@@ -388,6 +403,12 @@ void VPPredicator::convertPhisToBlends(VPBasicBlock *VPBB) {
InValEdgesMap[Val].push_back(Edge);
auto InValEdges = InValEdgesMap.takeVector();
+ if (InValEdges.size() == 1) {
+ PhiR->replaceAllUsesWith(InValEdges[0].first);
+ PhiR->eraseFromParent();
+ continue;
+ }
+
// Sort the incoming value order to match PhiR as much as possible.
llvm::stable_sort(InValEdges, [&PhiR](auto &L, auto &R) {
auto InVs = PhiR->incoming_values();
diff --git a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
index ba76b60903444..fa7291e1bd83e 100644
--- a/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/VPlan/predicator.ll
@@ -665,8 +665,6 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): merge.a
; CHECK-EMPTY:
; CHECK-NEXT: merge.a:
-; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = not ir<%c0>
-; CHECK-NEXT: BLEND ir<%blend.a> = ir<%v1>/ir<%c0> ir<%v1>/vp<[[VP5]]>
; CHECK-NEXT: EMIT ir<%d0> = icmp sgt ir<%iv>, ir<0>
; CHECK-NEXT: Successor(s): if.b
; CHECK-EMPTY:
@@ -675,16 +673,14 @@ define void @blend_chain_non_trivial(ptr noalias %a, ptr noalias %b) {
; CHECK-NEXT: Successor(s): if.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: if.b.inner:
-; CHECK-NEXT: EMIT vp<[[VP6:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
+; CHECK-NEXT: EMIT vp<[[VP5:%[0-9]+]]> = logical-and ir<%d0>, ir<%cb>
; CHECK-NEXT: Successor(s): merge.b.inner
; CHECK-EMPTY:
; CHECK-NEXT: merge.b.inner:
; CHECK-NEXT: Successor(s): merge.b
; CHECK-EMPTY:
; CHECK-NEXT: merge.b:
-; CHECK-NEXT: EMIT vp<[[VP7:%[0-9]+]]> = not ir<%d0>
-; CHECK-NEXT: BLEND ir<%blend.b> = ir<%v2>/ir<%d0> ir<%v2>/vp<[[VP7]]>
-; CHECK-NEXT: EMIT ir<%sum> = add ir<%blend.a>, ir<%blend.b>
+; CHECK-NEXT: EMIT ir<%sum> = add ir<%v1>, ir<%v2>
; CHECK-NEXT: EMIT store ir<%sum>, ir<%gep>
; CHECK-NEXT: Successor(s): loop.latch
; CHECK-EMPTY:
@@ -869,7 +865,7 @@ define void @look_thru_phi(i1 %c1, i1 %c2, i32 %x, i32 %y, ptr %p) {
; CHECK-NEXT: Successor(s): latch
; CHECK-EMPTY:
; CHECK-NEXT: latch:
-; CHECK-NEXT: BLEND ir<%phi> = ir<%phi1>/vp<[[VP7]]> ir<%x>/vp<[[VP8]]>
+; CHECK-NEXT: BLEND ir<%phi> = ir<%x>/ir<%c1> ir<%y>/vp<[[VP4]]>
; CHECK-NEXT: EMIT ir<%gep> = getelementptr ir<%p>, ir<%iv>
; CHECK-NEXT: EMIT store ir<%phi>, ir<%gep>
; CHECK-NEXT: EMIT ir<%iv.next> = add ir<%iv>, ir<1>
diff --git a/llvm/test/Transforms/LoopVectorize/predicator.ll b/llvm/test/Transforms/LoopVectorize/predicator.ll
index a03855edd8de7..98529cb1c227d 100644
--- a/llvm/test/Transforms/LoopVectorize/predicator.ll
+++ b/llvm/test/Transforms/LoopVectorize/predicator.ll
@@ -311,15 +311,7 @@ define void @look_thru_phi(i1 %c1, i1 %c2, i32 %x, i32 %y, ptr %p) {
; CHECK-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT]], <4 x i32> poison, <4 x i32> zeroinitializer
; CHECK-NEXT: [[BROADCAST_SPLATINSERT1:%.*]] = insertelement <4 x i32> poison, i32 [[X]], i64 0
; CHECK-NEXT: [[BROADCAST_SPLAT2:%.*]] = shufflevector <4 x i32> [[BROADCAST_SPLATINSERT1]], <4 x i32> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT3:%.*]] = insertelement <4 x i1> poison, i1 [[C2]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT4:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT3]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[BROADCAST_SPLATINSERT5:%.*]] = insertelement <4 x i1> poison, i1 [[C1]], i64 0
-; CHECK-NEXT: [[BROADCAST_SPLAT6:%.*]] = shufflevector <4 x i1> [[BROADCAST_SPLATINSERT5]], <4 x i1> poison, <4 x i32> zeroinitializer
-; CHECK-NEXT: [[TMP0:%.*]] = xor <4 x i1> [[BROADCAST_SPLAT4]], splat (i1 true)
-; CHECK-NEXT: [[TMP1:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[TMP0]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[PREDPHI:%.*]] = select <4 x i1> [[TMP1]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[BROADCAST_SPLAT]]
-; CHECK-NEXT: [[TMP2:%.*]] = select <4 x i1> [[BROADCAST_SPLAT6]], <4 x i1> [[BROADCAST_SPLAT4]], <4 x i1> zeroinitializer
-; CHECK-NEXT: [[PREDPHI7:%.*]] = select <4 x i1> [[TMP2]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[PREDPHI]]
+; CHECK-NEXT: [[PREDPHI7:%.*]] = select i1 [[C1]], <4 x i32> [[BROADCAST_SPLAT2]], <4 x i32> [[BROADCAST_SPLAT]]
; CHECK-NEXT: br label %[[VECTOR_BODY:.*]]
; CHECK: [[VECTOR_BODY]]:
; CHECK-NEXT: [[INDEX:%.*]] = phi i32 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[VECTOR_BODY]] ]
More information about the llvm-commits
mailing list