[llvm] [VPlan] Cost scalar IV steps in replicate regions. (PR #221995)
Florian Hahn via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 06:39:54 PDT 2026
https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/221995
>From f29b50fd57890ac533f70fb8ba03effd8f689f2e Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sat, 5 Sep 2026 19:36:38 +0100
Subject: [PATCH] [VPlan] Cost scalar IV steps in replicate regions.
Replace the return 0 bail-out in VPScalarIVStepsRecipe::computeCost in
replicate regions by properly scaling by the execution probability of
the region.
VPlan-based estimates of execution probabilities depend on
https://github.com/llvm/llvm-project/pull/216172 (include in PR)
---
.../lib/Transforms/Vectorize/VPlanRecipes.cpp | 16 ++++---
.../AArch64/induction-costs-sve.ll | 48 +------------------
.../AArch64/scalar-steps-cost.ll | 4 +-
.../X86/CostModel/store-scalarization-cost.ll | 4 +-
4 files changed, 15 insertions(+), 57 deletions(-)
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index fce56b42de84c..c0beb2e825e52 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -3144,11 +3144,6 @@ InstructionCost VPScalarIVStepsRecipe::computeCost(ElementCount VF,
if (!BaseIVTy->isIntegerTy())
return 0;
- // TODO: Add support for predicated regions. Requires scaling the cost by the
- // probability of entering the block.
- if (getRegion() && getRegion()->isReplicator())
- return 0;
-
// If only the first lane is used, then there won't be any code that remains
// in the loop for the first unrolled part.
if (vputils::onlyFirstLaneUsed(this))
@@ -3178,8 +3173,15 @@ InstructionCost VPScalarIVStepsRecipe::computeCost(ElementCount VF,
// %gep2 = getelementptr i8, ptr %base_gep, i32 1
// Therefore, in reality the cost is somewhere betwen 1*AddCost and
// (NumLanes - 1) * AddCost. For now, assume the cost of a single add.
- return Ctx.TTI.getArithmeticInstrCost(Instruction::Add, BaseIVTy,
- Ctx.CostKind);
+ //
+ // If the steps are generated inside a replicate region, scale by execution
+ // probability.
+ InstructionCost Cost =
+ Ctx.TTI.getArithmeticInstrCost(Instruction::Add, BaseIVTy, Ctx.CostKind);
+ const VPRegionBlock *Region = getRegion();
+ if (Region && Region->isReplicator())
+ Cost /= Ctx.getReplicateRegionCostDivisor(Region);
+ return Cost;
}
void VPScalarIVStepsRecipe::execute(VPTransformState &State) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs-sve.ll b/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs-sve.ll
index 7c3f140edf859..a99bd18aea6a9 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs-sve.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs-sve.ll
@@ -696,51 +696,7 @@ define void @exit_cond_zext_iv_store64(ptr %dst, i64 %N) {
;
; PRED-LABEL: define void @exit_cond_zext_iv_store64(
; PRED-SAME: ptr [[DST:%.*]], i64 [[N:%.*]]) {
-; PRED-NEXT: [[ENTRY:.*:]]
-; PRED-NEXT: [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
-; PRED-NEXT: br label %[[VECTOR_SCEVCHECK:.*]]
-; PRED: [[VECTOR_SCEVCHECK]]:
-; PRED-NEXT: [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[N]], i64 1)
-; PRED-NEXT: [[TMP2:%.*]] = trunc i64 [[TMP1]] to i32
-; PRED-NEXT: [[TMP3:%.*]] = add i32 1, [[TMP2]]
-; PRED-NEXT: [[TMP4:%.*]] = icmp ult i32 [[TMP3]], 1
-; PRED-NEXT: [[TMP5:%.*]] = icmp ugt i64 [[TMP1]], 4294967295
-; PRED-NEXT: [[TMP6:%.*]] = or i1 [[TMP4]], [[TMP5]]
-; PRED-NEXT: br i1 [[TMP6]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
-; PRED: [[VECTOR_PH]]:
-; PRED-NEXT: [[N_RND_UP:%.*]] = add i64 [[TMP0]], 1
-; PRED-NEXT: [[TMP7:%.*]] = and i64 [[N_RND_UP]], 1
-; PRED-NEXT: [[N_VEC:%.*]] = sub i64 [[N_RND_UP]], [[TMP7]]
-; PRED-NEXT: [[TRIP_COUNT_MINUS_1:%.*]] = sub i64 [[TMP0]], 1
-; PRED-NEXT: [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x i64> poison, i64 [[TRIP_COUNT_MINUS_1]], i64 0
-; PRED-NEXT: [[BROADCAST_SPLAT:%.*]] = shufflevector <2 x i64> [[BROADCAST_SPLATINSERT]], <2 x i64> poison, <2 x i32> zeroinitializer
-; PRED-NEXT: br label %[[VECTOR_BODY:.*]]
-; PRED: [[VECTOR_BODY]]:
-; PRED-NEXT: [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE2:.*]] ]
-; PRED-NEXT: [[VEC_IND:%.*]] = phi <2 x i64> [ <i64 0, i64 1>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[PRED_STORE_CONTINUE2]] ]
-; PRED-NEXT: [[TMP8:%.*]] = icmp ule <2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
-; PRED-NEXT: [[TMP9:%.*]] = extractelement <2 x i1> [[TMP8]], i64 0
-; PRED-NEXT: br i1 [[TMP9]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
-; PRED: [[PRED_STORE_IF]]:
-; PRED-NEXT: [[TMP10:%.*]] = getelementptr i64, ptr [[DST]], i64 [[INDEX]]
-; PRED-NEXT: store i64 0, ptr [[TMP10]], align 8
-; PRED-NEXT: br label %[[PRED_STORE_CONTINUE]]
-; PRED: [[PRED_STORE_CONTINUE]]:
-; PRED-NEXT: [[TMP11:%.*]] = extractelement <2 x i1> [[TMP8]], i64 1
-; PRED-NEXT: br i1 [[TMP11]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2]]
-; PRED: [[PRED_STORE_IF1]]:
-; PRED-NEXT: [[TMP12:%.*]] = add i64 [[INDEX]], 1
-; PRED-NEXT: [[TMP13:%.*]] = getelementptr i64, ptr [[DST]], i64 [[TMP12]]
-; PRED-NEXT: store i64 0, ptr [[TMP13]], align 8
-; PRED-NEXT: br label %[[PRED_STORE_CONTINUE2]]
-; PRED: [[PRED_STORE_CONTINUE2]]:
-; PRED-NEXT: [[INDEX_NEXT]] = add i64 [[INDEX]], 2
-; PRED-NEXT: [[VEC_IND_NEXT]] = add nuw <2 x i64> [[VEC_IND]], splat (i64 2)
-; PRED-NEXT: [[TMP14:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; PRED-NEXT: br i1 [[TMP14]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
-; PRED: [[MIDDLE_BLOCK]]:
-; PRED-NEXT: br label %[[EXIT:.*]]
-; PRED: [[SCALAR_PH]]:
+; PRED-NEXT: [[SCALAR_PH:.*]]:
; PRED-NEXT: br label %[[LOOP:.*]]
; PRED: [[LOOP]]:
; PRED-NEXT: [[IV_1:%.*]] = phi i32 [ 0, %[[SCALAR_PH]] ], [ [[IV_1_NEXT:%.*]], %[[LOOP]] ]
@@ -750,7 +706,7 @@ define void @exit_cond_zext_iv_store64(ptr %dst, i64 %N) {
; PRED-NEXT: [[IV_1_NEXT]] = add i32 [[IV_1]], 1
; PRED-NEXT: [[IV_EXT]] = zext i32 [[IV_1_NEXT]] to i64
; PRED-NEXT: [[C:%.*]] = icmp ult i64 [[IV_EXT]], [[N]]
-; PRED-NEXT: br i1 [[C]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP7:![0-9]+]]
+; PRED-NEXT: br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
; PRED: [[EXIT]]:
; PRED-NEXT: ret void
;
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/scalar-steps-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/scalar-steps-cost.ll
index 00b50c5d327b6..87b2c7b007d22 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/scalar-steps-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/scalar-steps-cost.ll
@@ -77,9 +77,9 @@ exit:
define void @scalar_steps_in_replicate_region(ptr noalias %dst, ptr noalias %src, i64 %n) {
; CHECK-LABEL: LV: Checking a loop in 'scalar_steps_in_replicate_region'
; CHECK: Cost of 0 for VF 2: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
-; CHECK: Cost of 0 for VF 2: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
-; CHECK: Cost of 0 for VF 4: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
+; CHECK: Cost of 0.5 for VF 2: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
; CHECK: Cost of 0 for VF 4: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
+; CHECK: Cost of 0.5 for VF 4: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
entry:
br label %loop
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/store-scalarization-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/store-scalarization-cost.ll
index a5ce45952aaad..bacd6aa7e2a88 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/store-scalarization-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/store-scalarization-cost.ll
@@ -50,12 +50,12 @@ define void @varying_pred_store_kept_in_replicate_region(ptr noalias %dst, ptr n
; CHECK: Cost of 3.5 for VF 2: profitable to scalarize %mul = mul i32 %l, %b
; CHECK: Cost of 0 for VF 2: REPLICATE ir<%mul> = mul ir<%l>, ir<%b>
; CHECK: Cost of 0 for VF 2: REPLICATE store ir<%mul>, ir<%gep.dst>
-; CHECK: Cost for VF 2: 9.5 (Estimated cost per lane: 4.5)
+; CHECK: Cost for VF 2: 10 (Estimated cost per lane: 5)
; CHECK: Cost of 2 for VF 4: profitable to scalarize store i32 %mul, ptr %gep.dst, align 8
; CHECK: Cost of 8.5 for VF 4: profitable to scalarize %mul = mul i32 %l, %b
; CHECK: Cost of 0 for VF 4: REPLICATE ir<%mul> = mul ir<%l>, ir<%b>
; CHECK: Cost of 0 for VF 4: REPLICATE store ir<%mul>, ir<%gep.dst>
-; CHECK: Cost for VF 4: 15.5 (Estimated cost per lane: 3.75)
+; CHECK: Cost for VF 4: 16 (Estimated cost per lane: 4)
; CHECK: LV: Selecting VF: 4.
;
entry:
More information about the llvm-commits
mailing list