[llvm] [LV] Print more debug to account for mismatch in final costs (PR #227757)
David Sherwood via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 08:52:28 PDT 2026
https://github.com/david-arm updated https://github.com/llvm/llvm-project/pull/227757
>From 0fd5bcb7cca20b311b4937db62bc3b81f71104fb Mon Sep 17 00:00:00 2001
From: David Sherwood <david.sherwood at arm.com>
Date: Wed, 30 Sep 2026 15:35:41 +0000
Subject: [PATCH 1/2] [LV] Print more debug to account for mismatch in final
costs
After printing out the costs of recipes we then print out
the total cost, including the cost per lane. Unfortunately,
the final cost often doesn't match the total of all the
recipe costs due to extra precomputed costs and reg spill
costs. This PR adds the missing information.
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 10 +++++++---
.../LoopVectorize/AArch64/maxbandwidth-regpressure.ll | 4 ++++
.../LoopVectorize/AArch64/predication_costs.ll | 6 ++++++
3 files changed, 17 insertions(+), 3 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 0986d8ea9e59d..6be37fedf266b 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5504,7 +5504,7 @@ bool VPCostContext::executesAtMostOnce(const VPlan &Plan, ElementCount VF) {
InstructionCost
LoopVectorizationPlanner::precomputeCosts(VPlan &Plan, ElementCount VF,
VPCostContext &CostCtx) const {
- InstructionCost Cost;
+ InstructionCost Cost = 0;
// If the vector loop gets executed exactly once with the given VF, ignore the
// costs of comparison and induction instructions, as they'll get simplified
@@ -5600,13 +5600,17 @@ InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
VPCostContext CostCtx(*TLI, Plan, *CM, Config,
/*ReusePrintingSlotTracker=*/true);
InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
+ LLVM_DEBUG(dbgs() << "Precomputed costs for VF " << VF << ": " << Cost << '\n');
// Now compute and add the VPlan-based cost.
Cost += Plan.cost(VF, CostCtx);
// Add the cost of spills due to excess register usage
- if (RU && Config.shouldConsiderRegPressureForVF(VF))
- Cost += RU->spillCost(TTI, Config.CostKind, ForceTargetNumVectorRegs);
+ if (RU && Config.shouldConsiderRegPressureForVF(VF)) {
+ InstructionCost SpillCost = RU->spillCost(TTI, Config.CostKind, ForceTargetNumVectorRegs);
+ LLVM_DEBUG(dbgs() << "Spill costs for VF " << VF << ": " << SpillCost << '\n');
+ Cost += SpillCost;
+ }
#ifndef NDEBUG
unsigned EstimatedWidth =
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/maxbandwidth-regpressure.ll b/llvm/test/Transforms/LoopVectorize/AArch64/maxbandwidth-regpressure.ll
index 3d5945d70768c..2e7eea7e65df2 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/maxbandwidth-regpressure.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/maxbandwidth-regpressure.ll
@@ -22,8 +22,10 @@ define i32 @dotp(ptr %a, ptr %b) #0 {
;
; CHECK-NOREGS-VP: Cost for VF vscale x 4: 6 (Estimated cost per lane: 1.
; CHECK-NOREGS-VP: LV(REG): Cost of 4 from 2 spills of Generic::VectorRC
+; CHECK-NOREGS-VP-NEXT: Spill costs for VF vscale x 8: 8
; CHECK-NOREGS-VP-NEXT: Cost for VF vscale x 8: 13 (Estimated cost per lane: 1.
; CHECK-NOREGS-VP: LV(REG): Cost of 4 from 2 spills of Generic::VectorRC
+; CHECK-NOREGS-VP-NEXT: Spill costs for VF vscale x 16: 8
; CHECK-NOREGS-VP-NEXT: Cost for VF vscale x 16: 13 (Estimated cost per lane: 0.
; CHECK-NOREGS-VP: LV: Selecting VF: vscale x 16.
entry:
@@ -90,8 +92,10 @@ define void @high_pressure(ptr %a, ptr %b) #0 {
; CHECK-NOREGS-VP: Cost for VF vscale x 4: 6 (Estimated cost per lane: 1.
; CHECK-NOREGS-VP: LV(REG): Cost of 6 from 3 spills of Generic::VectorRC
+; CHECK-NOREGS-VP-NEXT: Spill costs for VF vscale x 8: 10
; CHECK-NOREGS-VP-NEXT: Cost for VF vscale x 8: 20 (Estimated cost per lane: 2.
; CHECK-NOREGS-VP: LV(REG): Cost of 14 from 7 spills of Generic::VectorRC
+; CHECK-NOREGS-VP-NEXT: Spill costs for VF vscale x 16: 18
; CHECK-NOREGS-VP-NEXT: Cost for VF vscale x 16: 39 (Estimated cost per lane: 2.
; CHECK-NOREGS-VP: LV: Selecting VF: vscale x 4.
entry:
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/predication_costs.ll b/llvm/test/Transforms/LoopVectorize/AArch64/predication_costs.ll
index 940416c04324d..632e82413bedf 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/predication_costs.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/predication_costs.ll
@@ -20,6 +20,7 @@ target triple = "aarch64--linux-gnu"
;
; CHECK: Scalarizing and predicating: %tmp4 = udiv i32 %tmp2, %tmp3
; CHECK: Cost of 7 for VF 2: REPLICATE ir<%tmp4> = udiv ir<%tmp2>, ir<%tmp3> (S->V)
+; CHECK: Precomputed costs for VF 2: 4
;
define i32 @predicated_udiv(ptr %a, ptr %b, i1 %c, i64 %n) {
entry:
@@ -61,6 +62,7 @@ for.end:
;
; CHECK: Scalarizing and predicating: store i32 %tmp2, ptr %tmp0, align 4
; CHECK: Cost of 4 for VF 2: profitable to scalarize store i32 %tmp2, ptr %tmp0, align 4
+; CHECK: Precomputed costs for VF 2: 8
;
define void @predicated_store(ptr %a, i1 %c, i32 %x, i64 %n) {
entry:
@@ -95,6 +97,7 @@ for.end:
; CHECK: Scalarizing and predicating: store i32 %tmp2, ptr %addr, align 4
; CHECK: Cost of 0 for VF 2: forced scalar %addr = phi ptr [ %a, %entry ], [ %addr.next, %for.inc ]
; CHECK: Cost of 4 for VF 2: profitable to scalarize store i32 %tmp2, ptr %addr, align 4
+; CHECK: Precomputed costs for VF 2: 8
;
define void @predicated_store_phi(ptr %a, i1 %c, i32 %x, i64 %n) {
entry:
@@ -137,6 +140,7 @@ for.end:
; CHECK: Scalarizing and predicating: %tmp4 = udiv i32 %tmp2, %tmp3
; CHECK: Cost of 3 for VF 2: profitable to scalarize %tmp3 = add nsw i32 %tmp2, %x
; CHECK: Cost of 5 for VF 2: REPLICATE ir<%tmp4> = udiv ir<%tmp2>, ir<%tmp3> (S->V)
+; CHECK: Precomputed costs for VF 2: 7
;
define i32 @predicated_udiv_scalarized_operand(ptr %a, i1 %c, i32 %x, i64 %n) {
@@ -183,6 +187,7 @@ for.end:
; CHECK: Scalarizing: %tmp2 = add nsw i32 %tmp1, %x
; CHECK: Cost of 2 for VF 2: profitable to scalarize store i32 %tmp2, ptr %tmp0, align 4
; CHECK: Cost of 3 for VF 2: profitable to scalarize %tmp2 = add nsw i32 %tmp1, %x
+; CHECK: Precomputed costs for VF 2: 9
;
define void @predicated_store_scalarized_operand(ptr %a, i1 %c, i32 %x, i64 %n) {
entry:
@@ -239,6 +244,7 @@ for.end:
; CHECK: Cost of 1 for VF 2: WIDEN ir<%tmp2> = add ir<%tmp1>, ir<%x>
; CHECK: Cost of 7 for VF 2: REPLICATE ir<%tmp3> = sdiv ir<%tmp1>, ir<%tmp2>
; CHECK: Cost of 5 for VF 2: REPLICATE ir<%tmp4> = udiv ir<%tmp3>, ir<%tmp2>
+; CHECK: Precomputed costs for VF 2: 9
;
define void @predication_multi_context(ptr %a, i1 %c, i32 %x, i64 %n) {
entry:
>From a8b122b7b8ebb2c2aefb18850239fb26c3da8b6b Mon Sep 17 00:00:00 2001
From: David Sherwood <david.sherwood at arm.com>
Date: Wed, 30 Sep 2026 15:51:21 +0000
Subject: [PATCH 2/2] Fix formatting and remove pointless initialiser
---
llvm/lib/Transforms/Vectorize/LoopVectorize.cpp | 11 +++++++----
1 file changed, 7 insertions(+), 4 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 6be37fedf266b..6f657bf270fbe 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -5504,7 +5504,7 @@ bool VPCostContext::executesAtMostOnce(const VPlan &Plan, ElementCount VF) {
InstructionCost
LoopVectorizationPlanner::precomputeCosts(VPlan &Plan, ElementCount VF,
VPCostContext &CostCtx) const {
- InstructionCost Cost = 0;
+ InstructionCost Cost;
// If the vector loop gets executed exactly once with the given VF, ignore the
// costs of comparison and induction instructions, as they'll get simplified
@@ -5600,15 +5600,18 @@ InstructionCost LoopVectorizationPlanner::cost(VPlan &Plan, ElementCount VF,
VPCostContext CostCtx(*TLI, Plan, *CM, Config,
/*ReusePrintingSlotTracker=*/true);
InstructionCost Cost = precomputeCosts(Plan, VF, CostCtx);
- LLVM_DEBUG(dbgs() << "Precomputed costs for VF " << VF << ": " << Cost << '\n');
+ LLVM_DEBUG(dbgs() << "Precomputed costs for VF " << VF << ": " << Cost
+ << '\n');
// Now compute and add the VPlan-based cost.
Cost += Plan.cost(VF, CostCtx);
// Add the cost of spills due to excess register usage
if (RU && Config.shouldConsiderRegPressureForVF(VF)) {
- InstructionCost SpillCost = RU->spillCost(TTI, Config.CostKind, ForceTargetNumVectorRegs);
- LLVM_DEBUG(dbgs() << "Spill costs for VF " << VF << ": " << SpillCost << '\n');
+ InstructionCost SpillCost =
+ RU->spillCost(TTI, Config.CostKind, ForceTargetNumVectorRegs);
+ LLVM_DEBUG(dbgs() << "Spill costs for VF " << VF << ": " << SpillCost
+ << '\n');
Cost += SpillCost;
}
More information about the llvm-commits
mailing list