[llvm] [LV] Fix the cost of first order recurrence splice (PR #192473)

Sander de Smalen via llvm-commits llvm-commits at lists.llvm.org
Thu Apr 16 08:21:12 PDT 2026


https://github.com/sdesmalen-arm created https://github.com/llvm/llvm-project/pull/192473

The index had the wrong sign (for splice.right, the sign is negative), which meant that it calculates the cost of a splice.left operation. For SVE this makes a difference because a splice.left is lowered using an unpredicated EXT instruction, whereas a splice.right is lowered using a predicated SPLICE instruction, which needs a slightly higher cost.

The change in `reduction-recurrence-costs-sve.ll` happens because the vector loop is now less profitable (higher cost) and therefore requires a higher trip-count to be profitable (hence the extra umax).

>From 73a2ca8146527f39233a67920829ccaf32d21121 Mon Sep 17 00:00:00 2001
From: Sander de Smalen <sander.desmalen at arm.com>
Date: Thu, 16 Apr 2026 13:27:22 +0000
Subject: [PATCH 1/2] Pre-commit test

---
 .../LoopVectorize/AArch64/splice-cost.ll      | 31 +++++++++++++++++++
 1 file changed, 31 insertions(+)
 create mode 100644 llvm/test/Transforms/LoopVectorize/AArch64/splice-cost.ll

diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/splice-cost.ll b/llvm/test/Transforms/LoopVectorize/AArch64/splice-cost.ll
new file mode 100644
index 0000000000000..f99e61d70404a
--- /dev/null
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/splice-cost.ll
@@ -0,0 +1,31 @@
+; RUN: opt < %s -passes=loop-vectorize -disable-output -debug-only=loop-vectorize 2>&1 | FileCheck %s
+; REQUIRES: asserts
+target triple = "aarch64"
+
+; CHECK: Cost of 1 for VF 2: EMIT vp<{{.*}}> = first-order splice
+; CHECK: Cost of 3 for VF vscale x 2: EMIT vp<{{.*}}> = first-order splice
+
+define void @foo(ptr noalias %in, ptr noalias %out, i64 %n) "target-features"="+sve" {
+entry:
+  %load.prev = load i64, ptr %in, align 8
+  br label %for.body
+
+for.body:
+  %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %for.body ]
+  %prev = phi i64 [ %load.prev, %entry ], [ %load.cur, %for.body ]
+  %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+  %arrayidx.load = getelementptr inbounds nuw i64, ptr %in, i64 %indvars.iv.next
+  %load.cur = load i64, ptr %arrayidx.load
+  %add = add i64 %load.cur, %prev
+  %arrayidx.store = getelementptr inbounds nuw i64, ptr %out, i64 %indvars.iv
+  store i64 %add, ptr %arrayidx.store
+  %exitcond.not = icmp eq i64 %indvars.iv.next, %n
+  br i1 %exitcond.not, label %for.cond.cleanup, label %for.body, !llvm.loop !0
+
+for.cond.cleanup:
+  ret void
+}
+
+!0 = distinct !{!0, !1, !2}
+!1 = !{!"llvm.loop.mustprogress"}
+!2 = !{!"llvm.loop.unroll.disable"}

>From 3db137936956fad5b2fab9772952a163f81ee368 Mon Sep 17 00:00:00 2001
From: Sander de Smalen <sander.desmalen at arm.com>
Date: Thu, 16 Apr 2026 14:54:33 +0000
Subject: [PATCH 2/2] [LV] Fix the cost of first order recurrence splice

The index had the wrong sign (for splice.right, the sign is negative),
which meant that it calculates the cost of a splice.left operation.
For SVE this makes a difference because a splice.left is lowered
using an unpredicated EXT instruction, whereas a splice.right is
lowered using a predicated SPLICE instruction, which needs a slightly
higher cost.

The change in `reduction-recurrence-costs-sve.ll` happens because
the vector loop is now less profitable (higher cost) and therefore
requires a higher trip-count to be profitable (hence the extra umax).
---
 llvm/lib/Transforms/Vectorize/LoopVectorize.cpp            | 6 ++----
 llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp             | 7 ++-----
 .../AArch64/reduction-recurrence-costs-sve.ll              | 3 ++-
 3 files changed, 6 insertions(+), 10 deletions(-)

diff --git a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
index 3bf3f599c9828..4ad4731b59be6 100644
--- a/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/llvm/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -6035,12 +6035,10 @@ LoopVectorizationCostModel::getInstructionCost(Instruction *I,
 
     // First-order recurrences are replaced by vector shuffles inside the loop.
     if (VF.isVector() && Legal->isFixedOrderRecurrence(Phi)) {
-      SmallVector<int> Mask(VF.getKnownMinValue());
-      std::iota(Mask.begin(), Mask.end(), VF.getKnownMinValue() - 1);
       return TTI.getShuffleCost(TargetTransformInfo::SK_Splice,
                                 cast<VectorType>(VectorTy),
-                                cast<VectorType>(VectorTy), Mask, CostKind,
-                                VF.getKnownMinValue() - 1);
+                                cast<VectorType>(VectorTy), {}, CostKind,
+                                -(VF.getKnownMinValue() - 1));
     }
 
     // Phi nodes in non-header blocks (not inductions, reductions, etc.) are
diff --git a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
index cbb44198fe362..24aef9766f3b9 100644
--- a/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
+++ b/llvm/lib/Transforms/Vectorize/VPlanRecipes.cpp
@@ -1218,14 +1218,11 @@ InstructionCost VPInstruction::computeCost(ElementCount VF,
   }
   case VPInstruction::FirstOrderRecurrenceSplice: {
     assert(VF.isVector() && "Scalar FirstOrderRecurrenceSplice?");
-    SmallVector<int> Mask(VF.getKnownMinValue());
-    std::iota(Mask.begin(), Mask.end(), VF.getKnownMinValue() - 1);
     Type *VectorTy = toVectorTy(Ctx.Types.inferScalarType(this), VF);
-
     return Ctx.TTI.getShuffleCost(TargetTransformInfo::SK_Splice,
                                   cast<VectorType>(VectorTy),
-                                  cast<VectorType>(VectorTy), Mask,
-                                  Ctx.CostKind, VF.getKnownMinValue() - 1);
+                                  cast<VectorType>(VectorTy), {},
+                                  Ctx.CostKind, -(VF.getKnownMinValue() - 1));
   }
   case VPInstruction::ActiveLaneMask: {
     Type *ArgTy = Ctx.Types.inferScalarType(getOperand(0));
diff --git a/llvm/test/Transforms/LoopVectorize/AArch64/reduction-recurrence-costs-sve.ll b/llvm/test/Transforms/LoopVectorize/AArch64/reduction-recurrence-costs-sve.ll
index 8179adfec8201..3c88b9213c9a8 100644
--- a/llvm/test/Transforms/LoopVectorize/AArch64/reduction-recurrence-costs-sve.ll
+++ b/llvm/test/Transforms/LoopVectorize/AArch64/reduction-recurrence-costs-sve.ll
@@ -52,7 +52,8 @@ define i32 @chained_recurrences(i32 %x, i64 %y, ptr %src.1, i32 %z, ptr %src.2)
 ; VSCALEFORTUNING2-NEXT:    [[TMP0:%.*]] = add i64 [[Y]], 1
 ; VSCALEFORTUNING2-NEXT:    [[TMP1:%.*]] = call i64 @llvm.vscale.i64()
 ; VSCALEFORTUNING2-NEXT:    [[TMP2:%.*]] = shl nuw i64 [[TMP1]], 3
-; VSCALEFORTUNING2-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[TMP2]]
+; VSCALEFORTUNING2-NEXT:    [[UMAX:%.*]] = call i64 @llvm.umax.i64(i64 [[TMP2]], i64 16)
+; VSCALEFORTUNING2-NEXT:    [[MIN_ITERS_CHECK:%.*]] = icmp ult i64 [[TMP0]], [[UMAX]]
 ; VSCALEFORTUNING2-NEXT:    br i1 [[MIN_ITERS_CHECK]], label %[[SCALAR_PH:.*]], label %[[VECTOR_PH:.*]]
 ; VSCALEFORTUNING2:       [[VECTOR_PH]]:
 ; VSCALEFORTUNING2-NEXT:    [[TMP3:%.*]] = call i64 @llvm.vscale.i64()



More information about the llvm-commits mailing list