[llvm] Patch tryCanonicalizeStructToVector to handle split slice tails (PR #201434)

Yonah Goldberg via llvm-commits llvm-commits at lists.llvm.org
Wed Jun 3 12:08:32 PDT 2026


https://github.com/YonahGoldberg created https://github.com/llvm/llvm-project/pull/201434

We choose a vector alloca over a struct alloca when all users of the alloca are memory or lifetime intrinsics. But we only accounted for slices that start in the corresponding partition. We have to also check that all split slice tails overlapping the partition are memory or lifetime intrinsics

>From a83c4902c5ac50b1f5b17ba908e69db25be0be78 Mon Sep 17 00:00:00 2001
From: Yonah Goldberg <ygoldberg at nvidia.com>
Date: Wed, 3 Jun 2026 19:01:15 +0000
Subject: [PATCH] bug fix

---
 llvm/lib/Transforms/Scalar/SROA.cpp           | 19 +++++++---
 .../SROA/struct-to-vector-subpartition.ll     | 38 +++++++++++++++++++
 2 files changed, 51 insertions(+), 6 deletions(-)

diff --git a/llvm/lib/Transforms/Scalar/SROA.cpp b/llvm/lib/Transforms/Scalar/SROA.cpp
index 811dd373eca94..69865936001a7 100644
--- a/llvm/lib/Transforms/Scalar/SROA.cpp
+++ b/llvm/lib/Transforms/Scalar/SROA.cpp
@@ -5131,20 +5131,27 @@ static FixedVectorType *tryCanonicalizeStructToVector(StructType *STy,
   if (StructSize != VectorSize)
     return nullptr;
 
-  for (const Slice &S : P) {
+  auto IsMemIntrinsicOnlySlice = [](const Slice &S) {
     if (S.isDead())
-      continue;
+      return true;
     auto *U = S.getUse();
     if (!U)
-      continue;
+      return true;
 
     User *Usr = U->getUser();
     if (isa<LifetimeIntrinsic>(Usr) || isa<DbgInfoIntrinsic>(Usr))
-      continue;
+      return true;
 
-    if (!isa<MemIntrinsic>(Usr))
+    return isa<MemIntrinsic>(Usr);
+  };
+
+  for (const Slice &S : P)
+    if (!IsMemIntrinsicOnlySlice(S))
+      return nullptr;
+
+  for (const Slice *S : P.splitSliceTails())
+    if (!IsMemIntrinsicOnlySlice(*S))
       return nullptr;
-  }
 
   return VTy;
 }
diff --git a/llvm/test/Transforms/SROA/struct-to-vector-subpartition.ll b/llvm/test/Transforms/SROA/struct-to-vector-subpartition.ll
index 9edb8492aa460..c1d6ca989cd9c 100644
--- a/llvm/test/Transforms/SROA/struct-to-vector-subpartition.ll
+++ b/llvm/test/Transforms/SROA/struct-to-vector-subpartition.ll
@@ -67,3 +67,41 @@ merge:
   call void @llvm.memcpy.p0.p0.i64(ptr align 8 %dst, ptr align 8 %sel, i64 16, i1 false)
   ret void
 }
+
+; SROA sees these slices:
+;   [0,8)   load ptr
+;   [0,32)  store i256, splittable
+;   [8,16)  load i64, splittable
+;   [16,32) memcpy source, splittable
+;
+; These form three partitions:
+;   [0,8)   contains the ptr load and the store i256 slice that starts at 0
+;   [8,16)  contains the i64 load, plus the store i256 split tail
+;   [16,32) contains the memcpy source, plus the store i256 split tail
+;
+; The [16,32) subpartition has type { i64, i64 }, and the only slice that
+; starts in the partition is a memcpy. However, the whole-alloca i256 store is a
+; split tail overlapping the subpartition, so it must block struct-to-vector
+; fallback canonicalization.
+
+; CHECK-LABEL: define void @test_split_tail_store_blocks_subpartition_type(
+; CHECK-NOT: <2 x i64>
+; CHECK: ret void
+define void @test_split_tail_store_blocks_subpartition_type(ptr %dst, i256 %x) {
+entry:
+  %a = alloca { ptr, i64, i64, i64 }, align 8
+  store i256 %x, ptr %a, align 8
+
+  ; Force earlier partition boundaries so [16,32) is selected as a subpartition
+  ; whose type from getTypePartition is { i64, i64 }.
+  %p = load ptr, ptr %a, align 8
+  %puse = ptrtoint ptr %p to i64
+  %gep.a.8 = getelementptr inbounds i8, ptr %a, i64 8
+  %v8 = load i64, ptr %gep.a.8, align 8
+  %use = add i64 %puse, %v8
+  store i64 %use, ptr %dst, align 8
+
+  %gep.a.16 = getelementptr inbounds i8, ptr %a, i64 16
+  call void @llvm.memcpy.p0.p0.i64(ptr align 8 %dst, ptr align 8 %gep.a.16, i64 16, i1 false)
+  ret void
+}



More information about the llvm-commits mailing list