[llvm] [AArch64] Account for fixed-SVE high-lane insertelement/extractelement costs (PR #219244)
Utpal Bora via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 2 08:50:58 PDT 2026
================
@@ -259,3 +260,27 @@ entry:
store i8 %v1, ptr %out, align 1 ; Load has multiple uses
ret <8 x i8> %v2
}
+
+define void @fixed_sve_high_lane_insert_extract(<8 x float> %vf32, <4 x double> %vf64, float %f, double %d) vscale_range(2, 0) {
+; VLS-HIGH-LANE-LABEL: Printing analysis 'Cost Model Analysis' for function 'fixed_sve_high_lane_insert_extract':
+; VLS-HIGH-LANE-NEXT: Cost Model: Found costs of RThru:2 CodeSize:1 Lat:2 SizeLat:2 for: %ef32_low = extractelement <8 x float> %vf32, i32 3
+; VLS-HIGH-LANE-NEXT: Cost Model: Found costs of RThru:5 CodeSize:1 Lat:5 SizeLat:5 for: %ef32_high = extractelement <8 x float> %vf32, i32 4
+; VLS-HIGH-LANE-NEXT: Cost Model: Found costs of RThru:2 CodeSize:1 Lat:2 SizeLat:2 for: %ef64_low = extractelement <4 x double> %vf64, i32 1
+; VLS-HIGH-LANE-NEXT: Cost Model: Found costs of RThru:5 CodeSize:1 Lat:5 SizeLat:5 for: %ef64_high = extractelement <4 x double> %vf64, i32 2
+; VLS-HIGH-LANE-NEXT: Cost Model: Found costs of RThru:2 CodeSize:1 Lat:2 SizeLat:2 for: %if32_low = insertelement <8 x float> %vf32, float %f, i32 3
+; VLS-HIGH-LANE-NEXT: Cost Model: Found costs of RThru:5 CodeSize:1 Lat:5 SizeLat:5 for: %if32_high = insertelement <8 x float> %vf32, float %f, i32 4
+; VLS-HIGH-LANE-NEXT: Cost Model: Found costs of RThru:2 CodeSize:1 Lat:2 SizeLat:2 for: %if64_low = insertelement <4 x double> %vf64, double %d, i32 1
+; VLS-HIGH-LANE-NEXT: Cost Model: Found costs of RThru:5 CodeSize:1 Lat:5 SizeLat:5 for: %if64_high = insertelement <4 x double> %vf64, double %d, i32 2
+; VLS-HIGH-LANE-NEXT: Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for: ret void
+;
+ %ef32_low = extractelement <8 x float> %vf32, i32 3
+ %ef32_high = extractelement <8 x float> %vf32, i32 4
+ %ef64_low = extractelement <4 x double> %vf64, i32 1
+ %ef64_high = extractelement <4 x double> %vf64, i32 2
+ %if32_low = insertelement <8 x float> %vf32, float %f, i32 3
+ %if32_high = insertelement <8 x float> %vf32, float %f, i32 4
+ %if64_low = insertelement <4 x double> %vf64, double %d, i32 1
+ %if64_high = insertelement <4 x double> %vf64, double %d, i32 2
+ ret void
+}
----------------
utpalbora wrote:
Done
https://github.com/llvm/llvm-project/pull/219244
More information about the llvm-commits
mailing list