[llvm] [AArch64] Account for fixed-SVE high-lane insertelement/extractelement costs (PR #219244)

Utpal Bora via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 2 08:50:58 PDT 2026


================
@@ -259,3 +260,27 @@ entry:
   store i8 %v1, ptr %out, align 1  ; Load has multiple uses
   ret <8 x i8> %v2
 }
+
+define void @fixed_sve_high_lane_insert_extract(<8 x float> %vf32, <4 x double> %vf64, float %f, double %d) vscale_range(2, 0) {
+; VLS-HIGH-LANE-LABEL: Printing analysis 'Cost Model Analysis' for function 'fixed_sve_high_lane_insert_extract':
+; VLS-HIGH-LANE-NEXT:  Cost Model: Found costs of RThru:2 CodeSize:1 Lat:2 SizeLat:2 for:   %ef32_low = extractelement <8 x float> %vf32, i32 3
+; VLS-HIGH-LANE-NEXT:  Cost Model: Found costs of RThru:5 CodeSize:1 Lat:5 SizeLat:5 for:   %ef32_high = extractelement <8 x float> %vf32, i32 4
+; VLS-HIGH-LANE-NEXT:  Cost Model: Found costs of RThru:2 CodeSize:1 Lat:2 SizeLat:2 for:   %ef64_low = extractelement <4 x double> %vf64, i32 1
+; VLS-HIGH-LANE-NEXT:  Cost Model: Found costs of RThru:5 CodeSize:1 Lat:5 SizeLat:5 for:   %ef64_high = extractelement <4 x double> %vf64, i32 2
+; VLS-HIGH-LANE-NEXT:  Cost Model: Found costs of RThru:2 CodeSize:1 Lat:2 SizeLat:2 for:   %if32_low = insertelement <8 x float> %vf32, float %f, i32 3
+; VLS-HIGH-LANE-NEXT:  Cost Model: Found costs of RThru:5 CodeSize:1 Lat:5 SizeLat:5 for:   %if32_high = insertelement <8 x float> %vf32, float %f, i32 4
+; VLS-HIGH-LANE-NEXT:  Cost Model: Found costs of RThru:2 CodeSize:1 Lat:2 SizeLat:2 for:   %if64_low = insertelement <4 x double> %vf64, double %d, i32 1
+; VLS-HIGH-LANE-NEXT:  Cost Model: Found costs of RThru:5 CodeSize:1 Lat:5 SizeLat:5 for:   %if64_high = insertelement <4 x double> %vf64, double %d, i32 2
+; VLS-HIGH-LANE-NEXT:  Cost Model: Found costs of RThru:0 CodeSize:1 Lat:1 SizeLat:1 for:   ret void
+;
+  %ef32_low = extractelement <8 x float> %vf32, i32 3
+  %ef32_high = extractelement <8 x float> %vf32, i32 4
+  %ef64_low = extractelement <4 x double> %vf64, i32 1
+  %ef64_high = extractelement <4 x double> %vf64, i32 2
+  %if32_low = insertelement <8 x float> %vf32, float %f, i32 3
+  %if32_high = insertelement <8 x float> %vf32, float %f, i32 4
+  %if64_low = insertelement <4 x double> %vf64, double %d, i32 1
+  %if64_high = insertelement <4 x double> %vf64, double %d, i32 2
+  ret void
+}
----------------
utpalbora wrote:

Done

https://github.com/llvm/llvm-project/pull/219244


More information about the llvm-commits mailing list