[llvm] [SROA] Use SparseBitVector for splittable offsets (PR #222306)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 9 04:55:49 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-transforms

Author: John Brawn (john-brawn-arm)

<details>
<summary>Changes</summary>

Doing this means we consume memory proportional to the number of offsets, not the size of the alloca as before. Inverting the logic so that rather than initially marking all offsets as splittable then removing those that aren't, we instead initially have all offsets unsplittable and only add those that are splittable, means we should take time proportional to the number of slices, not to the combined size of slices.

This means we can remove the limit on alloca size, and this approach should also be faster.

---
Full diff: https://github.com/llvm/llvm-project/pull/222306.diff


2 Files Affected:

- (modified) llvm/lib/Transforms/Scalar/SROA.cpp (+41-36) 
- (added) llvm/test/Transforms/SROA/large-alloca.ll (+124) 


``````````diff
diff --git a/llvm/lib/Transforms/Scalar/SROA.cpp b/llvm/lib/Transforms/Scalar/SROA.cpp
index a0d7a0c921796..8a42edd33293f 100644
--- a/llvm/lib/Transforms/Scalar/SROA.cpp
+++ b/llvm/lib/Transforms/Scalar/SROA.cpp
@@ -30,9 +30,9 @@
 #include "llvm/ADT/PointerIntPair.h"
 #include "llvm/ADT/STLExtras.h"
 #include "llvm/ADT/SetVector.h"
-#include "llvm/ADT/SmallBitVector.h"
 #include "llvm/ADT/SmallPtrSet.h"
 #include "llvm/ADT/SmallVector.h"
+#include "llvm/ADT/SparseBitVector.h"
 #include "llvm/ADT/Statistic.h"
 #include "llvm/ADT/StringRef.h"
 #include "llvm/ADT/Twine.h"
@@ -5873,45 +5873,50 @@ bool SROA::splitAlloca(AllocaInst &AI, AllocaSlices &AS) {
   bool IsSorted = true;
 
   uint64_t AllocaSize = AI.getAllocationSize(DL)->getFixedValue();
-  const uint64_t MaxBitVectorSize = 1024;
-  if (AllocaSize <= MaxBitVectorSize) {
-    // If a byte boundary is included in any load or store, a slice starting or
-    // ending at the boundary is not splittable.
-    SmallBitVector SplittableOffset(AllocaSize + 1, true);
-    for (Slice &S : AS)
-      for (unsigned O = S.beginOffset() + 1;
-           O < S.endOffset() && O < AllocaSize; O++)
-        SplittableOffset.reset(O);
-
-    for (Slice &S : AS) {
-      if (!S.isSplittable())
-        continue;
-
-      if ((S.beginOffset() > AllocaSize || SplittableOffset[S.beginOffset()]) &&
-          (S.endOffset() > AllocaSize || SplittableOffset[S.endOffset()]))
-        continue;
-
-      if (isa<LoadInst>(S.getUse()->getUser()) ||
-          isa<StoreInst>(S.getUse()->getUser())) {
-        S.makeUnsplittable();
-        IsSorted = false;
+  // We can split at the begin and end offsets of each slice, but only if those
+  // offsets don't lie inside another slice. Because slices are ordered by
+  // increasing begin offset, and then decreasing end offset, we can consider
+  // the slices as being split up into sets with the same begin offset where we
+  // can ignore every slice except the first (the begin offset will already be
+  // handled as the begin offset of the set, and the end offset we know is not a
+  // splittable offset as it's inside the first slice of the set).
+  SparseBitVector<> SplittableOffset;
+  uint64_t CurBegin = 0, CurEnd = 0;
+  for (Slice &S : AS) {
+    // Check if we have a new set of slices
+    if (S.beginOffset() > CurBegin || S.endOffset() > CurEnd) {
+      // If the start isn't inside the previous set it's splittable
+      if (S.beginOffset() >= CurEnd) {
+        SplittableOffset.set(S.beginOffset());
+      }
+      // If the previous end is inside this slice then remove it
+      if (CurEnd > S.beginOffset() && CurEnd < S.endOffset()) {
+        SplittableOffset.reset(CurEnd);
+      }
+      CurBegin = S.beginOffset();
+      // If the end offset isn't inside the previous set it's splittable. We
+      // also don't update the end offset in that case, as the next set may also
+      // be inside the previous set.
+      if (S.endOffset() > CurEnd) {
+        CurEnd = S.endOffset();
+        SplittableOffset.set(CurEnd);
       }
     }
-  } else {
-    // We only allow whole-alloca splittable loads and stores
-    // for a large alloca to avoid creating too large BitVector.
-    for (Slice &S : AS) {
-      if (!S.isSplittable())
-        continue;
+  }
 
-      if (S.beginOffset() == 0 && S.endOffset() >= AllocaSize)
-        continue;
+  for (Slice &S : AS) {
+    if (!S.isSplittable())
+      continue;
 
-      if (isa<LoadInst>(S.getUse()->getUser()) ||
-          isa<StoreInst>(S.getUse()->getUser())) {
-        S.makeUnsplittable();
-        IsSorted = false;
-      }
+    if ((S.beginOffset() > AllocaSize ||
+         SplittableOffset.test(S.beginOffset())) &&
+        (S.endOffset() > AllocaSize || SplittableOffset.test(S.endOffset())))
+      continue;
+
+    if (isa<LoadInst>(S.getUse()->getUser()) ||
+        isa<StoreInst>(S.getUse()->getUser())) {
+      S.makeUnsplittable();
+      IsSorted = false;
     }
   }
 
diff --git a/llvm/test/Transforms/SROA/large-alloca.ll b/llvm/test/Transforms/SROA/large-alloca.ll
new file mode 100644
index 0000000000000..892d883e58c72
--- /dev/null
+++ b/llvm/test/Transforms/SROA/large-alloca.ll
@@ -0,0 +1,124 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=sroa,instcombine -S | FileCheck %s
+
+; Here we have an i16 stored inside a larger i64 load, which is then loaded by
+; an i32 load. We can split the i64 store store slice into two i32 halves, and
+; the upper half is then unused and so removed.
+
+define i32 @store_inside_load_low(i64 %x, i16 %y) {
+; CHECK-LABEL: define i32 @store_inside_load_low(
+; CHECK-SAME: i64 [[X:%.*]], i16 [[Y:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TMP0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT:%.*]] = and i32 [[TMP0]], -65536
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = zext i16 [[Y]] to i32
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i32 [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT:    ret i32 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+  %alloca = alloca [8 x i8], align 8
+  store i64 %x, ptr %alloca
+  store i16 %y, ptr %alloca
+  %load = load i32, ptr %alloca
+  ret i32 %load
+}
+
+define i32 @store_inside_load_high(i64 %x, i16 %y) {
+; CHECK-LABEL: define i32 @store_inside_load_high(
+; CHECK-SAME: i64 [[X:%.*]], i16 [[Y:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT:%.*]] = zext i16 [[Y]] to i32
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw i32 [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT]], 16
+; CHECK-NEXT:    [[TMP0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = and i32 [[TMP0]], 65535
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i32 [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT:    ret i32 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+  %alloca = alloca [8 x i8], align 8
+  %alloca.2 = getelementptr inbounds i8, ptr %alloca, i64 2
+  store i64 %x, ptr %alloca
+  store i16 %y, ptr %alloca.2
+  %load = load i32, ptr %alloca
+  ret i32 %load
+}
+
+; Having a larger alloca shouldn't change things.
+
+define i32 @store_inside_load_low_large_alloca(i64 %x, i16 %y) {
+; CHECK-LABEL: define i32 @store_inside_load_low_large_alloca(
+; CHECK-SAME: i64 [[X:%.*]], i16 [[Y:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TMP0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT:%.*]] = and i32 [[TMP0]], -65536
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = zext i16 [[Y]] to i32
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i32 [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT:    ret i32 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+  %alloca = alloca [2048 x i8], align 8
+  store i64 %x, ptr %alloca
+  store i16 %y, ptr %alloca
+  %load = load i32, ptr %alloca
+  ret i32 %load
+}
+
+define i32 @store_inside_load_high_large_alloca(i64 %x, i16 %y) {
+; CHECK-LABEL: define i32 @store_inside_load_high_large_alloca(
+; CHECK-SAME: i64 [[X:%.*]], i16 [[Y:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT:%.*]] = zext i16 [[Y]] to i32
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw i32 [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT]], 16
+; CHECK-NEXT:    [[TMP0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = and i32 [[TMP0]], 65535
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i32 [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT:    ret i32 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+  %alloca = alloca [2048 x i8], align 8
+  %alloca.2 = getelementptr inbounds i8, ptr %alloca, i64 2
+  store i64 %x, ptr %alloca
+  store i16 %y, ptr %alloca.2
+  %load = load i32, ptr %alloca
+  ret i32 %load
+}
+
+; Check that we get similar behavior using larger types.
+
+define i512 @store_inside_load_low_large_alloca_large_types(i1024 %x, i256 %y) {
+; CHECK-LABEL: define i512 @store_inside_load_low_large_alloca_large_types(
+; CHECK-SAME: i1024 [[X:%.*]], i256 [[Y:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[TMP0:%.*]] = trunc i1024 [[X]] to i512
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT:%.*]] = and i512 [[TMP0]], -115792089237316195423570985008687907853269984665640564039457584007913129639936
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = zext i256 [[Y]] to i512
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i512 [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT:    ret i512 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+  %alloca = alloca [2048 x i8], align 8
+  store i1024 %x, ptr %alloca
+  store i256 %y, ptr %alloca
+  %load = load i512, ptr %alloca
+  ret i512 %load
+}
+
+define i512 @store_inside_load_high_large_alloca_large_types(i1024 %x, i256 %y) {
+; CHECK-LABEL: define i512 @store_inside_load_high_large_alloca_large_types(
+; CHECK-SAME: i1024 [[X:%.*]], i256 [[Y:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*:]]
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT:%.*]] = zext i256 [[Y]] to i512
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw i512 [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT]], 256
+; CHECK-NEXT:    [[TMP0:%.*]] = trunc i1024 [[X]] to i512
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = and i512 [[TMP0]], 115792089237316195423570985008687907853269984665640564039457584007913129639935
+; CHECK-NEXT:    [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i512 [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT:    ret i512 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+  %alloca = alloca [2048 x i8], align 8
+  %alloca.32 = getelementptr inbounds i8, ptr %alloca, i64 32
+  store i1024 %x, ptr %alloca
+  store i256 %y, ptr %alloca.32
+  %load = load i512, ptr %alloca
+  ret i512 %load
+}

``````````

</details>


https://github.com/llvm/llvm-project/pull/222306


More information about the llvm-commits mailing list