[llvm] [AArch64] Increase fixed-length SVE scalarization costs (PR #226120)

via llvm-commits llvm-commits at lists.llvm.org
Thu Sep 24 04:17:46 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: Utpal Bora (utpalbora)

<details>
<summary>Changes</summary>

Scalarizing wide fixed-length SVE vectors may require expensive operand
materialization that the current cost model does not account for.

Added a conservative overhead to prevent SLP from selecting unprofitable trees.

---
Full diff: https://github.com/llvm/llvm-project/pull/226120.diff


3 Files Affected:

- (modified) llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp (+20-6) 
- (modified) llvm/test/Analysis/CostModel/AArch64/sve-vls-reduce-fp.ll (+9-9) 
- (added) llvm/test/Transforms/SLPVectorizer/AArch64/sve-wider-vector-inserts.ll (+71) 


``````````diff
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 581f37d4c17bc..21bf735451610 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -4891,12 +4891,26 @@ InstructionCost AArch64TTIImpl::getScalarizationOverhead(
     TTI::VectorInstrContext VIC) const {
   if (isa<ScalableVectorType>(Ty))
     return InstructionCost::getInvalid();
-  if (Ty->getElementType()->isFloatingPointTy())
-    return BaseT::getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
-                                           CostKind);
-  unsigned VecInstCost =
-      CostKind == TTI::TCK_CodeSize ? 1 : ST->getVectorInsertExtractBaseCost();
-  return DemandedElts.popcount() * (Insert + Extract) * VecInstCost;
+
+  std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
+  if (!ST->useSVEForFixedLengthVectors(LT.second)) {
+    if (Ty->getElementType()->isFloatingPointTy())
+      return BaseT::getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
+                                             CostKind);
+
+    unsigned VecInstCost = CostKind == TTI::TCK_CodeSize
+                               ? 1
+                               : ST->getVectorInsertExtractBaseCost();
+    return DemandedElts.popcount() * (Insert + Extract) * VecInstCost;
+  }
+
+  // Scalarizing fixed-length SVE vectors is expensive. Add an extra cost to
+  // prevent the SLP vectorizer from selecting unprofitable trees.
+  // TODO: Model the scalarization overhead of wide fixed-length SVE vectors
+  // accurately.
+  return BaseT::getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
+                                         CostKind) +
+         5;
 }
 
 std::optional<InstructionCost> AArch64TTIImpl::getFP16BF16PromoteCost(
diff --git a/llvm/test/Analysis/CostModel/AArch64/sve-vls-reduce-fp.ll b/llvm/test/Analysis/CostModel/AArch64/sve-vls-reduce-fp.ll
index ae3de7d1217dc..a6606a10b671a 100644
--- a/llvm/test/Analysis/CostModel/AArch64/sve-vls-reduce-fp.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/sve-vls-reduce-fp.ll
@@ -47,9 +47,9 @@ define void @reduce_fadd_strict_256b_types() {
 ; VSCALE-1-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 ; VSCALE-FROM-2-LABEL: 'reduce_fadd_strict_256b_types'
-; VSCALE-FROM-2-NEXT:  Cost Model: Found costs of RThru:62 CodeSize:47 Lat:94 SizeLat:62 for: %fadd_v16f16 = call half @llvm.vector.reduce.fadd.v16f16(half 0.000000e+00, <16 x half> poison)
-; VSCALE-FROM-2-NEXT:  Cost Model: Found costs of RThru:30 CodeSize:23 Lat:46 SizeLat:30 for: %fadd_v8f32 = call float @llvm.vector.reduce.fadd.v8f32(float 0.000000e+00, <8 x float> poison)
-; VSCALE-FROM-2-NEXT:  Cost Model: Found costs of RThru:14 CodeSize:11 Lat:22 SizeLat:14 for: %fadd_v4f64 = call double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> poison)
+; VSCALE-FROM-2-NEXT:  Cost Model: Found costs of RThru:67 CodeSize:52 Lat:99 SizeLat:67 for: %fadd_v16f16 = call half @llvm.vector.reduce.fadd.v16f16(half 0.000000e+00, <16 x half> poison)
+; VSCALE-FROM-2-NEXT:  Cost Model: Found costs of RThru:35 CodeSize:28 Lat:51 SizeLat:35 for: %fadd_v8f32 = call float @llvm.vector.reduce.fadd.v8f32(float 0.000000e+00, <8 x float> poison)
+; VSCALE-FROM-2-NEXT:  Cost Model: Found costs of RThru:19 CodeSize:16 Lat:27 SizeLat:19 for: %fadd_v4f64 = call double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> poison)
 ; VSCALE-FROM-2-NEXT:  Cost Model: Found costs of RThru:22 CodeSize:4 Lat:8 SizeLat:4 for: %fadd_v2f128 = call fp128 @llvm.vector.reduce.fadd.v2f128(fp128 poison, <2 x fp128> poison)
 ; VSCALE-FROM-2-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
@@ -69,16 +69,16 @@ define void @reduce_fadd_strict_512b_types() {
 ; VSCALE-1-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 ; VSCALE-2-LABEL: 'reduce_fadd_strict_512b_types'
-; VSCALE-2-NEXT:  Cost Model: Found costs of RThru:124 CodeSize:94 Lat:188 SizeLat:124 for: %fadd_v32f16 = call half @llvm.vector.reduce.fadd.v32f16(half 0.000000e+00, <32 x half> poison)
-; VSCALE-2-NEXT:  Cost Model: Found costs of RThru:60 CodeSize:46 Lat:92 SizeLat:60 for: %fadd_v16f32 = call float @llvm.vector.reduce.fadd.v16f32(float 0.000000e+00, <16 x float> poison)
-; VSCALE-2-NEXT:  Cost Model: Found costs of RThru:28 CodeSize:22 Lat:44 SizeLat:28 for: %fadd_v8f64 = call double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> poison)
+; VSCALE-2-NEXT:  Cost Model: Found costs of RThru:129 CodeSize:99 Lat:193 SizeLat:129 for: %fadd_v32f16 = call half @llvm.vector.reduce.fadd.v32f16(half 0.000000e+00, <32 x half> poison)
+; VSCALE-2-NEXT:  Cost Model: Found costs of RThru:65 CodeSize:51 Lat:97 SizeLat:65 for: %fadd_v16f32 = call float @llvm.vector.reduce.fadd.v16f32(float 0.000000e+00, <16 x float> poison)
+; VSCALE-2-NEXT:  Cost Model: Found costs of RThru:33 CodeSize:27 Lat:49 SizeLat:33 for: %fadd_v8f64 = call double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> poison)
 ; VSCALE-2-NEXT:  Cost Model: Found costs of RThru:44 CodeSize:8 Lat:16 SizeLat:8 for: %fadd_v4f128 = call fp128 @llvm.vector.reduce.fadd.v4f128(fp128 poison, <4 x fp128> poison)
 ; VSCALE-2-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
 ; VSCALE-FROM-4-LABEL: 'reduce_fadd_strict_512b_types'
-; VSCALE-FROM-4-NEXT:  Cost Model: Found costs of RThru:126 CodeSize:95 Lat:190 SizeLat:126 for: %fadd_v32f16 = call half @llvm.vector.reduce.fadd.v32f16(half 0.000000e+00, <32 x half> poison)
-; VSCALE-FROM-4-NEXT:  Cost Model: Found costs of RThru:62 CodeSize:47 Lat:94 SizeLat:62 for: %fadd_v16f32 = call float @llvm.vector.reduce.fadd.v16f32(float 0.000000e+00, <16 x float> poison)
-; VSCALE-FROM-4-NEXT:  Cost Model: Found costs of RThru:30 CodeSize:23 Lat:46 SizeLat:30 for: %fadd_v8f64 = call double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> poison)
+; VSCALE-FROM-4-NEXT:  Cost Model: Found costs of RThru:131 CodeSize:100 Lat:195 SizeLat:131 for: %fadd_v32f16 = call half @llvm.vector.reduce.fadd.v32f16(half 0.000000e+00, <32 x half> poison)
+; VSCALE-FROM-4-NEXT:  Cost Model: Found costs of RThru:67 CodeSize:52 Lat:99 SizeLat:67 for: %fadd_v16f32 = call float @llvm.vector.reduce.fadd.v16f32(float 0.000000e+00, <16 x float> poison)
+; VSCALE-FROM-4-NEXT:  Cost Model: Found costs of RThru:35 CodeSize:28 Lat:51 SizeLat:35 for: %fadd_v8f64 = call double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> poison)
 ; VSCALE-FROM-4-NEXT:  Cost Model: Found costs of RThru:44 CodeSize:8 Lat:16 SizeLat:8 for: %fadd_v4f128 = call fp128 @llvm.vector.reduce.fadd.v4f128(fp128 poison, <4 x fp128> poison)
 ; VSCALE-FROM-4-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
 ;
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/sve-wider-vector-inserts.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/sve-wider-vector-inserts.ll
new file mode 100644
index 0000000000000..9eb186675dcfb
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/sve-wider-vector-inserts.ll
@@ -0,0 +1,71 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -passes=slp-vectorizer -mtriple=aarch64-none-linux-gnu \
+; RUN:   -mattr=+sve -S %s | FileCheck %s
+
+; Check that the scalarization overhead for fixed-length SVE vectors prevents
+; SLP from selecting an unprofitable wide reduction tree.
+define double @insert_into_wider_vector(double %x, double %y) vscale_range(2, 0) {
+;
+; CHECK-LABEL: @insert_into_wider_vector(
+; CHECK-NEXT:    [[MUL0:%.*]] = fmul fast double [[X:%.*]], 0.000000e+00
+; CHECK-NEXT:    [[ADD0:%.*]] = fadd reassoc nsz double 0.000000e+00, [[MUL0]]
+; CHECK-NEXT:    [[MUL1:%.*]] = fmul fast double 0.000000e+00, [[Y:%.*]]
+; CHECK-NEXT:    [[TMP13:%.*]] = fadd reassoc nsz double [[ADD0]], [[MUL1]]
+; CHECK-NEXT:    [[TMP14:%.*]] = fmul fast double 1.000000e+00, [[X]]
+; CHECK-NEXT:    [[OP_RDX19:%.*]] = fadd reassoc nsz double [[TMP13]], [[TMP14]]
+; CHECK-NEXT:    [[MUL3:%.*]] = fmul fast double 0.000000e+00, 0.000000e+00
+; CHECK-NEXT:    [[ADD3:%.*]] = fadd reassoc nsz double [[OP_RDX19]], [[MUL3]]
+; CHECK-NEXT:    [[MUL4:%.*]] = fmul fast double 1.000000e+00, 1.000000e+00
+; CHECK-NEXT:    [[ADD4:%.*]] = fadd reassoc nsz double [[ADD3]], [[MUL4]]
+; CHECK-NEXT:    [[MUL5:%.*]] = fmul fast double [[X]], [[Y]]
+; CHECK-NEXT:    [[ADD5:%.*]] = fadd reassoc nsz double [[ADD4]], [[MUL5]]
+; CHECK-NEXT:    [[ADD6:%.*]] = fadd reassoc nsz double [[ADD5]], 0.000000e+00
+; CHECK-NEXT:    [[ADD7:%.*]] = fadd reassoc nsz double [[ADD6]], 0.000000e+00
+; CHECK-NEXT:    [[MUL8:%.*]] = fmul fast double [[X]], 0.000000e+00
+; CHECK-NEXT:    [[ADD8:%.*]] = fadd reassoc nsz double [[ADD7]], [[MUL8]]
+; CHECK-NEXT:    [[MUL9:%.*]] = fmul fast double 0.000000e+00, [[Y]]
+; CHECK-NEXT:    [[ADD9:%.*]] = fadd reassoc nsz double [[ADD8]], [[MUL9]]
+; CHECK-NEXT:    [[MUL10:%.*]] = fmul fast double 1.000000e+00, [[X]]
+; CHECK-NEXT:    [[ADD10:%.*]] = fadd reassoc nsz double [[ADD9]], [[MUL10]]
+; CHECK-NEXT:    [[ADD11:%.*]] = fadd reassoc nsz double [[ADD10]], 0.000000e+00
+; CHECK-NEXT:    [[ADD12:%.*]] = fadd reassoc nsz double [[ADD11]], 1.000000e+00
+; CHECK-NEXT:    [[MUL13:%.*]] = fmul fast double [[X]], [[Y]]
+; CHECK-NEXT:    [[ADD13:%.*]] = fadd reassoc nsz double [[ADD12]], [[MUL13]]
+; CHECK-NEXT:    [[ADD14:%.*]] = fadd reassoc nsz double [[ADD13]], 0.000000e+00
+; CHECK-NEXT:    [[OP_RDX20:%.*]] = fadd reassoc nsz double [[ADD14]], 0.000000e+00
+; CHECK-NEXT:    ret double [[OP_RDX20]]
+;
+  %mul0 = fmul fast double %x, 0.0
+  %add0 = fadd reassoc nsz double 0.0, %mul0
+  %mul1 = fmul fast double 0.0, %y
+  %add1 = fadd reassoc nsz double %add0, %mul1
+  %mul2 = fmul fast double 1.0, %x
+  %add2 = fadd reassoc nsz double %add1, %mul2
+  %mul3 = fmul fast double 0.0, 0.0
+  %add3 = fadd reassoc nsz double %add2, %mul3
+  %mul4 = fmul fast double 1.0, 1.0
+  %add4 = fadd reassoc nsz double %add3, %mul4
+  %mul5 = fmul fast double %x, %y
+  %add5 = fadd reassoc nsz double %add4, %mul5
+  %mul6 = fmul fast double 0.0, 0.0
+  %add6 = fadd reassoc nsz double %add5, %mul6
+  %mul7 = fmul fast double 0.0, 0.0
+  %add7 = fadd reassoc nsz double %add6, %mul7
+  %mul8 = fmul fast double %x, 0.0
+  %add8 = fadd reassoc nsz double %add7, %mul8
+  %mul9 = fmul fast double 0.0, %y
+  %add9 = fadd reassoc nsz double %add8, %mul9
+  %mul10 = fmul fast double 1.0, %x
+  %add10 = fadd reassoc nsz double %add9, %mul10
+  %mul11 = fmul fast double 0.0, 0.0
+  %add11 = fadd reassoc nsz double %add10, %mul11
+  %mul12 = fmul fast double 1.0, 1.0
+  %add12 = fadd reassoc nsz double %add11, %mul12
+  %mul13 = fmul fast double %x, %y
+  %add13 = fadd reassoc nsz double %add12, %mul13
+  %mul14 = fmul fast double 0.0, 0.0
+  %add14 = fadd reassoc nsz double %add13, %mul14
+  %mul15 = fmul fast double 0.0, 0.0
+  %add15 = fadd reassoc nsz double %add14, %mul15
+  ret double %add15
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/226120


More information about the llvm-commits mailing list