[llvm] [SLP] Split blocking build-vector stores in scalar chains (PR #194970)
Ryan Buchner via llvm-commits
llvm-commits at lists.llvm.org
Thu Apr 30 10:49:01 PDT 2026
================
@@ -0,0 +1,59 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt -passes=slp-vectorizer -S -mtriple=x86_64-unknown-linux-gnu < %s | FileCheck %s
+
+; The input has a scalar store chain with an explicit build-vector store in the
+; middle:
+;
+; p[0] = v0; p[1] = v1; p[2] = v2;
+; store <v3, v4, v5, v6> to &p[3];
+; p[7] = v7;
+;
+; Modeling the middle build-vector store as part of the surrounding store range
+; lets SLP rebuild the chain as two full <4 x float> stores instead of a mixed
+; <4 x float> plus <2 x float> / scalar tail shape.
+
+define void @buildvector_store_blocks_store_chain(ptr %p, float %a0, float %a1, float %a2, float %a3, float %a4, float %a5, float %a6, float %a7) {
+; CHECK-LABEL: define void @buildvector_store_blocks_store_chain(
+; CHECK-SAME: ptr [[P:%.*]], float [[A0:%.*]], float [[A1:%.*]], float [[A2:%.*]], float [[A3:%.*]], float [[A4:%.*]], float [[A5:%.*]], float [[A6:%.*]], float [[A7:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = insertelement <4 x float> poison, float [[A0]], i32 0
+; CHECK-NEXT: [[TMP1:%.*]] = insertelement <4 x float> [[TMP0]], float [[A1]], i32 1
+; CHECK-NEXT: [[TMP2:%.*]] = insertelement <4 x float> [[TMP1]], float [[A2]], i32 2
+; CHECK-NEXT: [[TMP3:%.*]] = insertelement <4 x float> [[TMP2]], float [[A3]], i32 3
+; CHECK-NEXT: [[TMP4:%.*]] = fadd <4 x float> [[TMP3]], splat (float 1.000000e+00)
+; CHECK-NEXT: [[TMP5:%.*]] = insertelement <4 x float> poison, float [[A4]], i32 0
+; CHECK-NEXT: [[TMP6:%.*]] = insertelement <4 x float> [[TMP5]], float [[A5]], i32 1
+; CHECK-NEXT: [[TMP7:%.*]] = insertelement <4 x float> [[TMP6]], float [[A6]], i32 2
+; CHECK-NEXT: [[TMP8:%.*]] = insertelement <4 x float> [[TMP7]], float [[A7]], i32 3
+; CHECK-NEXT: [[TMP9:%.*]] = fadd <4 x float> [[TMP8]], splat (float 1.000000e+00)
+; CHECK-NEXT: [[P3:%.*]] = getelementptr inbounds float, ptr [[P]], i64 3
+; CHECK-NEXT: [[TMP10:%.*]] = getelementptr i8, ptr [[P3]], i64 -12
+; CHECK-NEXT: store <4 x float> [[TMP4]], ptr [[TMP10]], align 4
+; CHECK-NEXT: [[P4:%.*]] = getelementptr i8, ptr [[P3]], i64 4
+; CHECK-NEXT: store <4 x float> [[TMP9]], ptr [[P4]], align 4
+; CHECK-NEXT: ret void
+;
+entry:
+ %v0 = fadd float %a0, 1.000000e+00
+ %v1 = fadd float %a1, 1.000000e+00
+ %v2 = fadd float %a2, 1.000000e+00
+ %v3 = fadd float %a3, 1.000000e+00
+ %v4 = fadd float %a4, 1.000000e+00
+ %v5 = fadd float %a5, 1.000000e+00
+ %v6 = fadd float %a6, 1.000000e+00
+ %v7 = fadd float %a7, 1.000000e+00
+ store float %v0, ptr %p, align 4
+ %p1 = getelementptr inbounds float, ptr %p, i64 1
+ store float %v1, ptr %p1, align 4
+ %p2 = getelementptr inbounds float, ptr %p, i64 2
+ store float %v2, ptr %p2, align 4
+ %p3 = getelementptr inbounds float, ptr %p, i64 3
+ %b0 = insertelement <4 x float> poison, float %v3, i32 0
+ %b1 = insertelement <4 x float> %b0, float %v4, i32 1
+ %b2 = insertelement <4 x float> %b1, float %v5, i32 2
+ %b3 = insertelement <4 x float> %b2, float %v6, i32 3
+ store <4 x float> %b3, ptr %p3, align 4
+ %p7 = getelementptr inbounds float, ptr %p, i64 7
+ store float %v7, ptr %p7, align 4
+ ret void
----------------
bababuck wrote:
I see, thank you!
https://github.com/llvm/llvm-project/pull/194970
More information about the llvm-commits
mailing list