[llvm] [VPlan] Cost scalar IV steps in replicate regions. (PR #221995)

Florian Hahn via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 9 02:48:32 PDT 2026


https://github.com/fhahn updated https://github.com/llvm/llvm-project/pull/221995

>From f29b50fd57890ac533f70fb8ba03effd8f689f2e Mon Sep 17 00:00:00 2001
From: Florian Hahn <flo at fhahn.com>
Date: Sat, 5 Sep 2026 19:36:38 +0100
Subject: [PATCH] [VPlan] Cost scalar IV steps in replicate regions.

Replace the return 0 bail-out in VPScalarIVStepsRecipe::computeCost in
replicate regions by properly scaling by the execution probability of
the region.

VPlan-based estimates of execution probabilities depend on
https://github.com/llvm/llvm-project/pull/216172 (include in PR)
---
 .../lib/Transforms/Vectorize/VPlanRecipes.cpp | 16 ++++---
 .../AArch64/induction-costs-sve.ll            | 48 +------------------
 .../AArch64/scalar-steps-cost.ll              |  4 +-
 .../X86/CostModel/store-scalarization-cost.ll |  4 +-
 4 files changed, 15 insertions(+), 57 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index fce56b42de84c..c0beb2e825e52 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -3144,11 +3144,6 @@ InstructionCost VPScalarIVStepsRecipe::computeCost(ElementCount VF,
   if (!BaseIVTy->isIntegerTy())
     return 0;
 
-  // TODO: Add support for predicated regions. Requires scaling the cost by the
-  // probability of entering the block.
-  if (getRegion() && getRegion()->isReplicator())
-    return 0;
-
   // If only the first lane is used, then there won't be any code that remains
   // in the loop for the first unrolled part.
   if (vputils::onlyFirstLaneUsed(this))
@@ -3178,8 +3173,15 @@ InstructionCost VPScalarIVStepsRecipe::computeCost(ElementCount VF,
   //  %gep2 = getelementptr i8, ptr %base_gep, i32 1
   // Therefore, in reality the cost is somewhere betwen 1*AddCost and
   // (NumLanes - 1) * AddCost. For now, assume the cost of a single add.
-  return Ctx.TTI.getArithmeticInstrCost(Instruction::Add, BaseIVTy,
-                                        Ctx.CostKind);
+  //
+  // If the steps are generated inside a replicate region, scale by execution
+  // probability.
+  InstructionCost Cost =
+      Ctx.TTI.getArithmeticInstrCost(Instruction::Add, BaseIVTy, Ctx.CostKind);
+  const VPRegionBlock *Region = getRegion();
+  if (Region && Region->isReplicator())
+    Cost /= Ctx.getReplicateRegionCostDivisor(Region);
+  return Cost;
 }
 
 void VPScalarIVStepsRecipe::execute(VPTransformState &State) {
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs-sve.ll b/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs-sve.ll
index 7c3f140edf859..a99bd18aea6a9 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs-sve.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/induction-costs-sve.ll
@@ -696,51 +696,7 @@ define void @exit_cond_zext_iv_store64(ptr %dst, i64 %N) {
 ;
 ; PRED-LABEL: define void @exit_cond_zext_iv_store64(
 ; PRED-SAME: ptr [[DST:%.*]], i64 [[N:%.*]]) {
-; PRED-NEXT:  [[ENTRY:.*:]]
-; PRED-NEXT:    [[TMP0:%.*]] = call i64 @llvm.umax.i64(i64 [[N]], i64 1)
-; PRED-NEXT:    br label %[[VECTOR_SCEVCHECK:.*]]
-; PRED:       [[VECTOR_SCEVCHECK]]:
-; PRED-NEXT:    [[TMP1:%.*]] = call i64 @llvm.usub.sat.i64(i64 [[N]], i64 1)
-; PRED-NEXT:    [[TMP2:%.*]] = trunc i64 [[TMP1]] to i32
-; PRED-NEXT:    [[TMP3:%.*]] = add i32 1, [[TMP2]]
-; PRED-NEXT:    [[TMP4:%.*]] = icmp ult i32 [[TMP3]], 1
-; PRED-NEXT:    [[TMP5:%.*]] = icmp ugt i64 [[TMP1]], 4294967295
-; PRED-NEXT:    [[TMP6:%.*]] = or i1 [[TMP4]], [[TMP5]]
-; PRED-NEXT:    br i1 [[TMP6]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
-; PRED:       [[VECTOR_PH]]:
-; PRED-NEXT:    [[N_RND_UP:%.*]] = add i64 [[TMP0]], 1
-; PRED-NEXT:    [[TMP7:%.*]] = and i64 [[N_RND_UP]], 1
-; PRED-NEXT:    [[N_VEC:%.*]] = sub i64 [[N_RND_UP]], [[TMP7]]
-; PRED-NEXT:    [[TRIP_COUNT_MINUS_1:%.*]] = sub i64 [[TMP0]], 1
-; PRED-NEXT:    [[BROADCAST_SPLATINSERT:%.*]] = insertelement <2 x i64> poison, i64 [[TRIP_COUNT_MINUS_1]], i64 0
-; PRED-NEXT:    [[BROADCAST_SPLAT:%.*]] = shufflevector <2 x i64> [[BROADCAST_SPLATINSERT]], <2 x i64> poison, <2 x i32> zeroinitializer
-; PRED-NEXT:    br label %[[VECTOR_BODY:.*]]
-; PRED:       [[VECTOR_BODY]]:
-; PRED-NEXT:    [[INDEX:%.*]] = phi i64 [ 0, %[[VECTOR_PH]] ], [ [[INDEX_NEXT:%.*]], %[[PRED_STORE_CONTINUE2:.*]] ]
-; PRED-NEXT:    [[VEC_IND:%.*]] = phi <2 x i64> [ <i64 0, i64 1>, %[[VECTOR_PH]] ], [ [[VEC_IND_NEXT:%.*]], %[[PRED_STORE_CONTINUE2]] ]
-; PRED-NEXT:    [[TMP8:%.*]] = icmp ule <2 x i64> [[VEC_IND]], [[BROADCAST_SPLAT]]
-; PRED-NEXT:    [[TMP9:%.*]] = extractelement <2 x i1> [[TMP8]], i64 0
-; PRED-NEXT:    br i1 [[TMP9]], label %[[PRED_STORE_IF:.*]], label %[[PRED_STORE_CONTINUE:.*]]
-; PRED:       [[PRED_STORE_IF]]:
-; PRED-NEXT:    [[TMP10:%.*]] = getelementptr i64, ptr [[DST]], i64 [[INDEX]]
-; PRED-NEXT:    store i64 0, ptr [[TMP10]], align 8
-; PRED-NEXT:    br label %[[PRED_STORE_CONTINUE]]
-; PRED:       [[PRED_STORE_CONTINUE]]:
-; PRED-NEXT:    [[TMP11:%.*]] = extractelement <2 x i1> [[TMP8]], i64 1
-; PRED-NEXT:    br i1 [[TMP11]], label %[[PRED_STORE_IF1:.*]], label %[[PRED_STORE_CONTINUE2]]
-; PRED:       [[PRED_STORE_IF1]]:
-; PRED-NEXT:    [[TMP12:%.*]] = add i64 [[INDEX]], 1
-; PRED-NEXT:    [[TMP13:%.*]] = getelementptr i64, ptr [[DST]], i64 [[TMP12]]
-; PRED-NEXT:    store i64 0, ptr [[TMP13]], align 8
-; PRED-NEXT:    br label %[[PRED_STORE_CONTINUE2]]
-; PRED:       [[PRED_STORE_CONTINUE2]]:
-; PRED-NEXT:    [[INDEX_NEXT]] = add i64 [[INDEX]], 2
-; PRED-NEXT:    [[VEC_IND_NEXT]] = add nuw <2 x i64> [[VEC_IND]], splat (i64 2)
-; PRED-NEXT:    [[TMP14:%.*]] = icmp eq i64 [[INDEX_NEXT]], [[N_VEC]]
-; PRED-NEXT:    br i1 [[TMP14]], label %[[MIDDLE_BLOCK:.*]], label %[[VECTOR_BODY]], !llvm.loop [[LOOP6:![0-9]+]]
-; PRED:       [[MIDDLE_BLOCK]]:
-; PRED-NEXT:    br label %[[EXIT:.*]]
-; PRED:       [[SCALAR_PH]]:
+; PRED-NEXT:  [[SCALAR_PH:.*]]:
 ; PRED-NEXT:    br label %[[LOOP:.*]]
 ; PRED:       [[LOOP]]:
 ; PRED-NEXT:    [[IV_1:%.*]] = phi i32 [ 0, %[[SCALAR_PH]] ], [ [[IV_1_NEXT:%.*]], %[[LOOP]] ]
@@ -750,7 +706,7 @@ define void @exit_cond_zext_iv_store64(ptr %dst, i64 %N) {
 ; PRED-NEXT:    [[IV_1_NEXT]] = add i32 [[IV_1]], 1
 ; PRED-NEXT:    [[IV_EXT]] = zext i32 [[IV_1_NEXT]] to i64
 ; PRED-NEXT:    [[C:%.*]] = icmp ult i64 [[IV_EXT]], [[N]]
-; PRED-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT]], !llvm.loop [[LOOP7:![0-9]+]]
+; PRED-NEXT:    br i1 [[C]], label %[[LOOP]], label %[[EXIT:.*]]
 ; PRED:       [[EXIT]]:
 ; PRED-NEXT:    ret void
 ;
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/scalar-steps-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/scalar-steps-cost.ll
index 00b50c5d327b6..87b2c7b007d22 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/scalar-steps-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/scalar-steps-cost.ll
@@ -77,9 +77,9 @@ exit:
 define void @scalar_steps_in_replicate_region(ptr noalias %dst, ptr noalias %src, i64 %n) {
 ; CHECK-LABEL: LV: Checking a loop in 'scalar_steps_in_replicate_region'
 ; CHECK: Cost of 0 for VF 2: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
-; CHECK: Cost of 0 for VF 2: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
-; CHECK: Cost of 0 for VF 4: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
+; CHECK: Cost of 0.5 for VF 2: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
 ; CHECK: Cost of 0 for VF 4: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
+; CHECK: Cost of 0.5 for VF 4: {{.*}} = SCALAR-STEPS {{.*}}, ir<1>, {{.*}}
 entry:
   br label %loop
 
diff --git a/llvm/test/Transforms/LoopVectorize/X86/CostModel/store-scalarization-cost.ll b/llvm/test/Transforms/LoopVectorize/X86/CostModel/store-scalarization-cost.ll
index a5ce45952aaad..bacd6aa7e2a88 100644
--- a/llvm/test/Transforms/LoopVectorize/X86/CostModel/store-scalarization-cost.ll
+++ b/llvm/test/Transforms/LoopVectorize/X86/CostModel/store-scalarization-cost.ll
@@ -50,12 +50,12 @@ define void @varying_pred_store_kept_in_replicate_region(ptr noalias %dst, ptr n
 ; CHECK:  Cost of 3.5 for VF 2: profitable to scalarize %mul = mul i32 %l, %b
 ; CHECK:  Cost of 0 for VF 2: REPLICATE ir<%mul> = mul ir<%l>, ir<%b>
 ; CHECK:  Cost of 0 for VF 2: REPLICATE store ir<%mul>, ir<%gep.dst>
-; CHECK:  Cost for VF 2: 9.5 (Estimated cost per lane: 4.5)
+; CHECK:  Cost for VF 2: 10 (Estimated cost per lane: 5)
 ; CHECK:  Cost of 2 for VF 4: profitable to scalarize store i32 %mul, ptr %gep.dst, align 8
 ; CHECK:  Cost of 8.5 for VF 4: profitable to scalarize %mul = mul i32 %l, %b
 ; CHECK:  Cost of 0 for VF 4: REPLICATE ir<%mul> = mul ir<%l>, ir<%b>
 ; CHECK:  Cost of 0 for VF 4: REPLICATE store ir<%mul>, ir<%gep.dst>
-; CHECK:  Cost for VF 4: 15.5 (Estimated cost per lane: 3.75)
+; CHECK:  Cost for VF 4: 16 (Estimated cost per lane: 4)
 ; CHECK:  LV: Selecting VF: 4.
 ;
 entry:



More information about the llvm-commits mailing list