[llvm] [AArch64] Increase fixed-length SVE scalarization costs (PR #226120)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 04:17:46 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-transforms
Author: Utpal Bora (utpalbora)
<details>
<summary>Changes</summary>
Scalarizing wide fixed-length SVE vectors may require expensive operand
materialization that the current cost model does not account for.
Added a conservative overhead to prevent SLP from selecting unprofitable trees.
---
Full diff: https://github.com/llvm/llvm-project/pull/226120.diff
3 Files Affected:
- (modified) llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp (+20-6)
- (modified) llvm/test/Analysis/CostModel/AArch64/sve-vls-reduce-fp.ll (+9-9)
- (added) llvm/test/Transforms/SLPVectorizer/AArch64/sve-wider-vector-inserts.ll (+71)
``````````diff
diff --git a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
index 581f37d4c17bc..21bf735451610 100644
--- a/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
+++ b/llvm/lib/Target/AArch64/AArch64TargetTransformInfo.cpp
@@ -4891,12 +4891,26 @@ InstructionCost AArch64TTIImpl::getScalarizationOverhead(
TTI::VectorInstrContext VIC) const {
if (isa<ScalableVectorType>(Ty))
return InstructionCost::getInvalid();
- if (Ty->getElementType()->isFloatingPointTy())
- return BaseT::getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
- CostKind);
- unsigned VecInstCost =
- CostKind == TTI::TCK_CodeSize ? 1 : ST->getVectorInsertExtractBaseCost();
- return DemandedElts.popcount() * (Insert + Extract) * VecInstCost;
+
+ std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
+ if (!ST->useSVEForFixedLengthVectors(LT.second)) {
+ if (Ty->getElementType()->isFloatingPointTy())
+ return BaseT::getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
+ CostKind);
+
+ unsigned VecInstCost = CostKind == TTI::TCK_CodeSize
+ ? 1
+ : ST->getVectorInsertExtractBaseCost();
+ return DemandedElts.popcount() * (Insert + Extract) * VecInstCost;
+ }
+
+ // Scalarizing fixed-length SVE vectors is expensive. Add an extra cost to
+ // prevent the SLP vectorizer from selecting unprofitable trees.
+ // TODO: Model the scalarization overhead of wide fixed-length SVE vectors
+ // accurately.
+ return BaseT::getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
+ CostKind) +
+ 5;
}
std::optional<InstructionCost> AArch64TTIImpl::getFP16BF16PromoteCost(
diff --git a/llvm/test/Analysis/CostModel/AArch64/sve-vls-reduce-fp.ll b/llvm/test/Analysis/CostModel/AArch64/sve-vls-reduce-fp.ll
index ae3de7d1217dc..a6606a10b671a 100644
--- a/llvm/test/Analysis/CostModel/AArch64/sve-vls-reduce-fp.ll
+++ b/llvm/test/Analysis/CostModel/AArch64/sve-vls-reduce-fp.ll
@@ -47,9 +47,9 @@ define void @reduce_fadd_strict_256b_types() {
; VSCALE-1-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; VSCALE-FROM-2-LABEL: 'reduce_fadd_strict_256b_types'
-; VSCALE-FROM-2-NEXT: Cost Model: Found costs of RThru:62 CodeSize:47 Lat:94 SizeLat:62 for: %fadd_v16f16 = call half @llvm.vector.reduce.fadd.v16f16(half 0.000000e+00, <16 x half> poison)
-; VSCALE-FROM-2-NEXT: Cost Model: Found costs of RThru:30 CodeSize:23 Lat:46 SizeLat:30 for: %fadd_v8f32 = call float @llvm.vector.reduce.fadd.v8f32(float 0.000000e+00, <8 x float> poison)
-; VSCALE-FROM-2-NEXT: Cost Model: Found costs of RThru:14 CodeSize:11 Lat:22 SizeLat:14 for: %fadd_v4f64 = call double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> poison)
+; VSCALE-FROM-2-NEXT: Cost Model: Found costs of RThru:67 CodeSize:52 Lat:99 SizeLat:67 for: %fadd_v16f16 = call half @llvm.vector.reduce.fadd.v16f16(half 0.000000e+00, <16 x half> poison)
+; VSCALE-FROM-2-NEXT: Cost Model: Found costs of RThru:35 CodeSize:28 Lat:51 SizeLat:35 for: %fadd_v8f32 = call float @llvm.vector.reduce.fadd.v8f32(float 0.000000e+00, <8 x float> poison)
+; VSCALE-FROM-2-NEXT: Cost Model: Found costs of RThru:19 CodeSize:16 Lat:27 SizeLat:19 for: %fadd_v4f64 = call double @llvm.vector.reduce.fadd.v4f64(double 0.000000e+00, <4 x double> poison)
; VSCALE-FROM-2-NEXT: Cost Model: Found costs of RThru:22 CodeSize:4 Lat:8 SizeLat:4 for: %fadd_v2f128 = call fp128 @llvm.vector.reduce.fadd.v2f128(fp128 poison, <2 x fp128> poison)
; VSCALE-FROM-2-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
@@ -69,16 +69,16 @@ define void @reduce_fadd_strict_512b_types() {
; VSCALE-1-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; VSCALE-2-LABEL: 'reduce_fadd_strict_512b_types'
-; VSCALE-2-NEXT: Cost Model: Found costs of RThru:124 CodeSize:94 Lat:188 SizeLat:124 for: %fadd_v32f16 = call half @llvm.vector.reduce.fadd.v32f16(half 0.000000e+00, <32 x half> poison)
-; VSCALE-2-NEXT: Cost Model: Found costs of RThru:60 CodeSize:46 Lat:92 SizeLat:60 for: %fadd_v16f32 = call float @llvm.vector.reduce.fadd.v16f32(float 0.000000e+00, <16 x float> poison)
-; VSCALE-2-NEXT: Cost Model: Found costs of RThru:28 CodeSize:22 Lat:44 SizeLat:28 for: %fadd_v8f64 = call double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> poison)
+; VSCALE-2-NEXT: Cost Model: Found costs of RThru:129 CodeSize:99 Lat:193 SizeLat:129 for: %fadd_v32f16 = call half @llvm.vector.reduce.fadd.v32f16(half 0.000000e+00, <32 x half> poison)
+; VSCALE-2-NEXT: Cost Model: Found costs of RThru:65 CodeSize:51 Lat:97 SizeLat:65 for: %fadd_v16f32 = call float @llvm.vector.reduce.fadd.v16f32(float 0.000000e+00, <16 x float> poison)
+; VSCALE-2-NEXT: Cost Model: Found costs of RThru:33 CodeSize:27 Lat:49 SizeLat:33 for: %fadd_v8f64 = call double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> poison)
; VSCALE-2-NEXT: Cost Model: Found costs of RThru:44 CodeSize:8 Lat:16 SizeLat:8 for: %fadd_v4f128 = call fp128 @llvm.vector.reduce.fadd.v4f128(fp128 poison, <4 x fp128> poison)
; VSCALE-2-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
; VSCALE-FROM-4-LABEL: 'reduce_fadd_strict_512b_types'
-; VSCALE-FROM-4-NEXT: Cost Model: Found costs of RThru:126 CodeSize:95 Lat:190 SizeLat:126 for: %fadd_v32f16 = call half @llvm.vector.reduce.fadd.v32f16(half 0.000000e+00, <32 x half> poison)
-; VSCALE-FROM-4-NEXT: Cost Model: Found costs of RThru:62 CodeSize:47 Lat:94 SizeLat:62 for: %fadd_v16f32 = call float @llvm.vector.reduce.fadd.v16f32(float 0.000000e+00, <16 x float> poison)
-; VSCALE-FROM-4-NEXT: Cost Model: Found costs of RThru:30 CodeSize:23 Lat:46 SizeLat:30 for: %fadd_v8f64 = call double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> poison)
+; VSCALE-FROM-4-NEXT: Cost Model: Found costs of RThru:131 CodeSize:100 Lat:195 SizeLat:131 for: %fadd_v32f16 = call half @llvm.vector.reduce.fadd.v32f16(half 0.000000e+00, <32 x half> poison)
+; VSCALE-FROM-4-NEXT: Cost Model: Found costs of RThru:67 CodeSize:52 Lat:99 SizeLat:67 for: %fadd_v16f32 = call float @llvm.vector.reduce.fadd.v16f32(float 0.000000e+00, <16 x float> poison)
+; VSCALE-FROM-4-NEXT: Cost Model: Found costs of RThru:35 CodeSize:28 Lat:51 SizeLat:35 for: %fadd_v8f64 = call double @llvm.vector.reduce.fadd.v8f64(double 0.000000e+00, <8 x double> poison)
; VSCALE-FROM-4-NEXT: Cost Model: Found costs of RThru:44 CodeSize:8 Lat:16 SizeLat:8 for: %fadd_v4f128 = call fp128 @llvm.vector.reduce.fadd.v4f128(fp128 poison, <4 x fp128> poison)
; VSCALE-FROM-4-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
;
diff --git a/llvm/test/Transforms/SLPVectorizer/AArch64/sve-wider-vector-inserts.ll b/llvm/test/Transforms/SLPVectorizer/AArch64/sve-wider-vector-inserts.ll
new file mode 100644
index 0000000000000..9eb186675dcfb
--- /dev/null
+++ b/llvm/test/Transforms/SLPVectorizer/AArch64/sve-wider-vector-inserts.ll
@@ -0,0 +1,71 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; RUN: opt -passes=slp-vectorizer -mtriple=aarch64-none-linux-gnu \
+; RUN: -mattr=+sve -S %s | FileCheck %s
+
+; Check that the scalarization overhead for fixed-length SVE vectors prevents
+; SLP from selecting an unprofitable wide reduction tree.
+define double @insert_into_wider_vector(double %x, double %y) vscale_range(2, 0) {
+;
+; CHECK-LABEL: @insert_into_wider_vector(
+; CHECK-NEXT: [[MUL0:%.*]] = fmul fast double [[X:%.*]], 0.000000e+00
+; CHECK-NEXT: [[ADD0:%.*]] = fadd reassoc nsz double 0.000000e+00, [[MUL0]]
+; CHECK-NEXT: [[MUL1:%.*]] = fmul fast double 0.000000e+00, [[Y:%.*]]
+; CHECK-NEXT: [[TMP13:%.*]] = fadd reassoc nsz double [[ADD0]], [[MUL1]]
+; CHECK-NEXT: [[TMP14:%.*]] = fmul fast double 1.000000e+00, [[X]]
+; CHECK-NEXT: [[OP_RDX19:%.*]] = fadd reassoc nsz double [[TMP13]], [[TMP14]]
+; CHECK-NEXT: [[MUL3:%.*]] = fmul fast double 0.000000e+00, 0.000000e+00
+; CHECK-NEXT: [[ADD3:%.*]] = fadd reassoc nsz double [[OP_RDX19]], [[MUL3]]
+; CHECK-NEXT: [[MUL4:%.*]] = fmul fast double 1.000000e+00, 1.000000e+00
+; CHECK-NEXT: [[ADD4:%.*]] = fadd reassoc nsz double [[ADD3]], [[MUL4]]
+; CHECK-NEXT: [[MUL5:%.*]] = fmul fast double [[X]], [[Y]]
+; CHECK-NEXT: [[ADD5:%.*]] = fadd reassoc nsz double [[ADD4]], [[MUL5]]
+; CHECK-NEXT: [[ADD6:%.*]] = fadd reassoc nsz double [[ADD5]], 0.000000e+00
+; CHECK-NEXT: [[ADD7:%.*]] = fadd reassoc nsz double [[ADD6]], 0.000000e+00
+; CHECK-NEXT: [[MUL8:%.*]] = fmul fast double [[X]], 0.000000e+00
+; CHECK-NEXT: [[ADD8:%.*]] = fadd reassoc nsz double [[ADD7]], [[MUL8]]
+; CHECK-NEXT: [[MUL9:%.*]] = fmul fast double 0.000000e+00, [[Y]]
+; CHECK-NEXT: [[ADD9:%.*]] = fadd reassoc nsz double [[ADD8]], [[MUL9]]
+; CHECK-NEXT: [[MUL10:%.*]] = fmul fast double 1.000000e+00, [[X]]
+; CHECK-NEXT: [[ADD10:%.*]] = fadd reassoc nsz double [[ADD9]], [[MUL10]]
+; CHECK-NEXT: [[ADD11:%.*]] = fadd reassoc nsz double [[ADD10]], 0.000000e+00
+; CHECK-NEXT: [[ADD12:%.*]] = fadd reassoc nsz double [[ADD11]], 1.000000e+00
+; CHECK-NEXT: [[MUL13:%.*]] = fmul fast double [[X]], [[Y]]
+; CHECK-NEXT: [[ADD13:%.*]] = fadd reassoc nsz double [[ADD12]], [[MUL13]]
+; CHECK-NEXT: [[ADD14:%.*]] = fadd reassoc nsz double [[ADD13]], 0.000000e+00
+; CHECK-NEXT: [[OP_RDX20:%.*]] = fadd reassoc nsz double [[ADD14]], 0.000000e+00
+; CHECK-NEXT: ret double [[OP_RDX20]]
+;
+ %mul0 = fmul fast double %x, 0.0
+ %add0 = fadd reassoc nsz double 0.0, %mul0
+ %mul1 = fmul fast double 0.0, %y
+ %add1 = fadd reassoc nsz double %add0, %mul1
+ %mul2 = fmul fast double 1.0, %x
+ %add2 = fadd reassoc nsz double %add1, %mul2
+ %mul3 = fmul fast double 0.0, 0.0
+ %add3 = fadd reassoc nsz double %add2, %mul3
+ %mul4 = fmul fast double 1.0, 1.0
+ %add4 = fadd reassoc nsz double %add3, %mul4
+ %mul5 = fmul fast double %x, %y
+ %add5 = fadd reassoc nsz double %add4, %mul5
+ %mul6 = fmul fast double 0.0, 0.0
+ %add6 = fadd reassoc nsz double %add5, %mul6
+ %mul7 = fmul fast double 0.0, 0.0
+ %add7 = fadd reassoc nsz double %add6, %mul7
+ %mul8 = fmul fast double %x, 0.0
+ %add8 = fadd reassoc nsz double %add7, %mul8
+ %mul9 = fmul fast double 0.0, %y
+ %add9 = fadd reassoc nsz double %add8, %mul9
+ %mul10 = fmul fast double 1.0, %x
+ %add10 = fadd reassoc nsz double %add9, %mul10
+ %mul11 = fmul fast double 0.0, 0.0
+ %add11 = fadd reassoc nsz double %add10, %mul11
+ %mul12 = fmul fast double 1.0, 1.0
+ %add12 = fadd reassoc nsz double %add11, %mul12
+ %mul13 = fmul fast double %x, %y
+ %add13 = fadd reassoc nsz double %add12, %mul13
+ %mul14 = fmul fast double 0.0, 0.0
+ %add14 = fadd reassoc nsz double %add13, %mul14
+ %mul15 = fmul fast double 0.0, 0.0
+ %add15 = fadd reassoc nsz double %add14, %mul15
+ ret double %add15
+}
``````````
</details>
https://github.com/llvm/llvm-project/pull/226120
More information about the llvm-commits
mailing list