[llvm] [VPlan] Pass the mask's compare predicate when costing a blend. (PR #218422)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Wed Aug 26 06:43:43 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/218422
>From 4586a4b03a70e993e4d89c1eae3f596aa5657044 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Thu, 20 Aug 2026 12:22:13 +0100
Subject: [PATCH 1/2] [VPlan] Pass the mask's compare predicate when costing a
blend.
Pass the predicate to getCmpSelInstrCost in VPBlendRecipe if the mask is
a compare. This improves cost estimates on targets like AArch64 where
selects on compares can be lowered more efficiently.
---
llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp | 13 ++++++++++---
.../Transforms/LoopVectorize/AArch64/cmp_cost.ll | 8 ++++----
2 files changed, 14 insertions(+), 7 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 306a4a259affe..8c0fb36684320 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -3334,9 +3334,16 @@ InstructionCost VPBlendRecipe::computeCost(ElementCount VF,
Type *ResultTy = toVectorTy(this->getScalarType(), VF);
Type *CmpTy = toVectorTy(Type::getInt1Ty(Ctx.LLVMCtx), VF);
- return (getNumIncomingValues() - 1) *
- Ctx.TTI.getCmpSelInstrCost(Instruction::Select, ResultTy, CmpTy,
- CmpInst::BAD_ICMP_PREDICATE, Ctx.CostKind);
+
+ InstructionCost Cost = 0;
+ for (unsigned I = 1, E = getNumIncomingValues(); I != E; ++I) {
+ CmpPredicate Pred;
+ if (!match(getMask(I), m_Cmp(Pred, m_VPValue(), m_VPValue())))
+ Pred = CmpInst::BAD_ICMP_PREDICATE;
+ Cost += Ctx.TTI.getCmpSelInstrCost(Instruction::Select, ResultTy, CmpTy,
+ Pred, Ctx.CostKind);
+ }
+ return Cost;
}
#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/cmp_cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/cmp_cost.ll
index 5bc97910eb2f0..e33ab81d36fa6 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/cmp_cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/cmp_cost.ll
@@ -183,7 +183,7 @@ define i32 @switch_to_cmp(ptr %s, ptr %dst, i64 %n) {
; CHECK: Cost of 0 for VF 2: EMIT vp<[[VP36:%[0-9]+]]> = or vp<[[VP35]]>, vp<[[VP21]]>
; CHECK: Cost of 1 for VF 2: WIDEN ir<%c.4> = add ir<%c>, ir<4>
; CHECK: Cost of 1 for VF 2: WIDEN ir<%c.1> = add ir<%c>, ir<1>
-; CHECK: Cost of 12 for VF 2: BLEND ir<%c.next> = ir<%c> ir<%c.4>/vp<[[VP22]]> ir<%c.1>/vp<[[VP36]]>
+; CHECK: Cost of 7 for VF 2: BLEND ir<%c.next> = ir<%c> ir<%c.4>/vp<[[VP22]]> ir<%c.1>/vp<[[VP36]]>
; CHECK: Cost of 0 for VF 2: CLONE ir<%dst.gep> = getelementptr ir<%dst>, vp<[[VP5]]>
; CHECK: Cost of 0 for VF 2: vp<[[VP37:%[0-9]+]]> = vector-pointer i8, ir<%dst.gep>, ir<1>
; CHECK: Cost of 4 for VF 2: WIDEN store vp<[[VP37]]>, ir<%l>
@@ -242,7 +242,7 @@ define i32 @switch_to_cmp(ptr %s, ptr %dst, i64 %n) {
; CHECK: Cost of 0 for VF 4: EMIT vp<[[VP36]]> = or vp<[[VP35]]>, vp<[[VP21]]>
; CHECK: Cost of 1 for VF 4: WIDEN ir<%c.4> = add ir<%c>, ir<4>
; CHECK: Cost of 1 for VF 4: WIDEN ir<%c.1> = add ir<%c>, ir<1>
-; CHECK: Cost of 24 for VF 4: BLEND ir<%c.next> = ir<%c> ir<%c.4>/vp<[[VP22]]> ir<%c.1>/vp<[[VP36]]>
+; CHECK: Cost of 13 for VF 4: BLEND ir<%c.next> = ir<%c> ir<%c.4>/vp<[[VP22]]> ir<%c.1>/vp<[[VP36]]>
; CHECK: Cost of 0 for VF 4: CLONE ir<%dst.gep> = getelementptr ir<%dst>, vp<[[VP5]]>
; CHECK: Cost of 0 for VF 4: vp<[[VP37]]> = vector-pointer i8, ir<%dst.gep>, ir<1>
; CHECK: Cost of 2 for VF 4: WIDEN store vp<[[VP37]]>, ir<%l>
@@ -301,7 +301,7 @@ define i32 @switch_to_cmp(ptr %s, ptr %dst, i64 %n) {
; CHECK: Cost of 0 for VF 8: EMIT vp<[[VP36]]> = or vp<[[VP35]]>, vp<[[VP21]]>
; CHECK: Cost of 2 for VF 8: WIDEN ir<%c.4> = add ir<%c>, ir<4>
; CHECK: Cost of 2 for VF 8: WIDEN ir<%c.1> = add ir<%c>, ir<1>
-; CHECK: Cost of 16 for VF 8: BLEND ir<%c.next> = ir<%c> ir<%c.4>/vp<[[VP22]]> ir<%c.1>/vp<[[VP36]]>
+; CHECK: Cost of 10 for VF 8: BLEND ir<%c.next> = ir<%c> ir<%c.4>/vp<[[VP22]]> ir<%c.1>/vp<[[VP36]]>
; CHECK: Cost of 0 for VF 8: CLONE ir<%dst.gep> = getelementptr ir<%dst>, vp<[[VP5]]>
; CHECK: Cost of 0 for VF 8: vp<[[VP37]]> = vector-pointer i8, ir<%dst.gep>, ir<1>
; CHECK: Cost of 1 for VF 8: WIDEN store vp<[[VP37]]>, ir<%l>
@@ -360,7 +360,7 @@ define i32 @switch_to_cmp(ptr %s, ptr %dst, i64 %n) {
; CHECK: Cost of 0 for VF 16: EMIT vp<[[VP36]]> = or vp<[[VP35]]>, vp<[[VP21]]>
; CHECK: Cost of 4 for VF 16: WIDEN ir<%c.4> = add ir<%c>, ir<4>
; CHECK: Cost of 4 for VF 16: WIDEN ir<%c.1> = add ir<%c>, ir<1>
-; CHECK: Cost of 32 for VF 16: BLEND ir<%c.next> = ir<%c> ir<%c.4>/vp<[[VP22]]> ir<%c.1>/vp<[[VP36]]>
+; CHECK: Cost of 20 for VF 16: BLEND ir<%c.next> = ir<%c> ir<%c.4>/vp<[[VP22]]> ir<%c.1>/vp<[[VP36]]>
; CHECK: Cost of 0 for VF 16: CLONE ir<%dst.gep> = getelementptr ir<%dst>, vp<[[VP5]]>
; CHECK: Cost of 0 for VF 16: vp<[[VP37]]> = vector-pointer i8, ir<%dst.gep>, ir<1>
; CHECK: Cost of 1 for VF 16: WIDEN store vp<[[VP37]]>, ir<%l>
>From e841a5770d8196b090b3871dea02ab9f92a11e11 Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Wed, 26 Aug 2026 14:43:11 +0100
Subject: [PATCH 2/2] !fixup use BAD_FCMPE_PREDICATE for FP types
---
llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp | 3 ++-
1 file changed, 2 insertions(+), 1 deletion(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index 6ee51a5e08eae..0cf6c66dc91a6 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -3339,7 +3339,8 @@ InstructionCost VPBlendRecipe::computeCost(ElementCount VF,
for (unsigned I = 1, E = getNumIncomingValues(); I != E; ++I) {
CmpPredicate Pred;
if (!match(getMask(I), m_Cmp(Pred, m_VPValue(), m_VPValue())))
- Pred = CmpInst::BAD_ICMP_PREDICATE;
+ Pred = getScalarType()->isFloatingPointTy() ? CmpInst::BAD_FCMP_PREDICATE
+ : CmpInst::BAD_ICMP_PREDICATE;
Cost += Ctx.TTI.getCmpSelInstrCost(Instruction::Select, ResultTy, CmpTy,
Pred, Ctx.CostKind);
}
More information about the llvm-commits
mailing list