[llvm] [SROA] Use SparseBitVector for splittable offsets (PR #222306)
John Brawn via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 04:53:41 PDT 2026
https://github.com/john-brawn-arm created https://github.com/llvm/llvm-project/pull/222306
Doing this means we consume memory proportional to the number of offsets, not the size of the alloca as before. Inverting the logic so that rather than initially marking all offsets as splittable then removing those that aren't, we instead initially have all offsets unsplittable and only add those that are splittable, means we should take time proportional to the number of slices, not to the combined size of slices.
This means we can remove the limit on alloca size, and this approach should also be faster.
>From 29cfac5b3a3fc9f6110843844c411dbaae6c8125 Mon Sep 17 00:00:00 2001
From: John Brawn <john.brawn at arm.com>
Date: Fri, 7 Aug 2026 12:30:22 +0100
Subject: [PATCH] [SROA] Use SparseBitVector for splittable offsets
Doing this means we consume memory proportional to the number of
offsets, not the size of the alloca as before. Inverting the logic so
that rather than initially marking all offsets as splittable then
removing those that aren't, we instead initially have all offsets
unsplittable and only add those that are splittable, means we should
take time proportional to the number of slices, not to the combined
size of slices.
This means we can remove the limit on alloca size, and this approach
should also be faster.
---
llvm/lib/Transforms/Scalar/SROA.cpp | 77 +++++++-------
llvm/test/Transforms/SROA/large-alloca.ll | 124 ++++++++++++++++++++++
2 files changed, 165 insertions(+), 36 deletions(-)
create mode 100644 llvm/test/Transforms/SROA/large-alloca.ll
diff --git a/llvm/lib/Transforms/Scalar/SROA.cpp b/llvm/lib/Transforms/Scalar/SROA.cpp
index a0d7a0c921796..8a42edd33293f 100644
--- a/llvm/lib/Transforms/Scalar/SROA.cpp
+++ b/llvm/lib/Transforms/Scalar/SROA.cpp
@@ -30,9 +30,9 @@
#include "llvm/ADT/PointerIntPair.h"
#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/SetVector.h"
-#include "llvm/ADT/SmallBitVector.h"
#include "llvm/ADT/SmallPtrSet.h"
#include "llvm/ADT/SmallVector.h"
+#include "llvm/ADT/SparseBitVector.h"
#include "llvm/ADT/Statistic.h"
#include "llvm/ADT/StringRef.h"
#include "llvm/ADT/Twine.h"
@@ -5873,45 +5873,50 @@ bool SROA::splitAlloca(AllocaInst &AI, AllocaSlices &AS) {
bool IsSorted = true;
uint64_t AllocaSize = AI.getAllocationSize(DL)->getFixedValue();
- const uint64_t MaxBitVectorSize = 1024;
- if (AllocaSize <= MaxBitVectorSize) {
- // If a byte boundary is included in any load or store, a slice starting or
- // ending at the boundary is not splittable.
- SmallBitVector SplittableOffset(AllocaSize + 1, true);
- for (Slice &S : AS)
- for (unsigned O = S.beginOffset() + 1;
- O < S.endOffset() && O < AllocaSize; O++)
- SplittableOffset.reset(O);
-
- for (Slice &S : AS) {
- if (!S.isSplittable())
- continue;
-
- if ((S.beginOffset() > AllocaSize || SplittableOffset[S.beginOffset()]) &&
- (S.endOffset() > AllocaSize || SplittableOffset[S.endOffset()]))
- continue;
-
- if (isa<LoadInst>(S.getUse()->getUser()) ||
- isa<StoreInst>(S.getUse()->getUser())) {
- S.makeUnsplittable();
- IsSorted = false;
+ // We can split at the begin and end offsets of each slice, but only if those
+ // offsets don't lie inside another slice. Because slices are ordered by
+ // increasing begin offset, and then decreasing end offset, we can consider
+ // the slices as being split up into sets with the same begin offset where we
+ // can ignore every slice except the first (the begin offset will already be
+ // handled as the begin offset of the set, and the end offset we know is not a
+ // splittable offset as it's inside the first slice of the set).
+ SparseBitVector<> SplittableOffset;
+ uint64_t CurBegin = 0, CurEnd = 0;
+ for (Slice &S : AS) {
+ // Check if we have a new set of slices
+ if (S.beginOffset() > CurBegin || S.endOffset() > CurEnd) {
+ // If the start isn't inside the previous set it's splittable
+ if (S.beginOffset() >= CurEnd) {
+ SplittableOffset.set(S.beginOffset());
+ }
+ // If the previous end is inside this slice then remove it
+ if (CurEnd > S.beginOffset() && CurEnd < S.endOffset()) {
+ SplittableOffset.reset(CurEnd);
+ }
+ CurBegin = S.beginOffset();
+ // If the end offset isn't inside the previous set it's splittable. We
+ // also don't update the end offset in that case, as the next set may also
+ // be inside the previous set.
+ if (S.endOffset() > CurEnd) {
+ CurEnd = S.endOffset();
+ SplittableOffset.set(CurEnd);
}
}
- } else {
- // We only allow whole-alloca splittable loads and stores
- // for a large alloca to avoid creating too large BitVector.
- for (Slice &S : AS) {
- if (!S.isSplittable())
- continue;
+ }
- if (S.beginOffset() == 0 && S.endOffset() >= AllocaSize)
- continue;
+ for (Slice &S : AS) {
+ if (!S.isSplittable())
+ continue;
- if (isa<LoadInst>(S.getUse()->getUser()) ||
- isa<StoreInst>(S.getUse()->getUser())) {
- S.makeUnsplittable();
- IsSorted = false;
- }
+ if ((S.beginOffset() > AllocaSize ||
+ SplittableOffset.test(S.beginOffset())) &&
+ (S.endOffset() > AllocaSize || SplittableOffset.test(S.endOffset())))
+ continue;
+
+ if (isa<LoadInst>(S.getUse()->getUser()) ||
+ isa<StoreInst>(S.getUse()->getUser())) {
+ S.makeUnsplittable();
+ IsSorted = false;
}
}
diff --git a/llvm/test/Transforms/SROA/large-alloca.ll b/llvm/test/Transforms/SROA/large-alloca.ll
new file mode 100644
index 0000000000000..892d883e58c72
--- /dev/null
+++ b/llvm/test/Transforms/SROA/large-alloca.ll
@@ -0,0 +1,124 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 6
+; RUN: opt < %s -passes=sroa,instcombine -S | FileCheck %s
+
+; Here we have an i16 stored inside a larger i64 load, which is then loaded by
+; an i32 load. We can split the i64 store store slice into two i32 halves, and
+; the upper half is then unused and so removed.
+
+define i32 @store_inside_load_low(i64 %x, i16 %y) {
+; CHECK-LABEL: define i32 @store_inside_load_low(
+; CHECK-SAME: i64 [[X:%.*]], i16 [[Y:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT:%.*]] = and i32 [[TMP0]], -65536
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = zext i16 [[Y]] to i32
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i32 [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT: ret i32 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+ %alloca = alloca [8 x i8], align 8
+ store i64 %x, ptr %alloca
+ store i16 %y, ptr %alloca
+ %load = load i32, ptr %alloca
+ ret i32 %load
+}
+
+define i32 @store_inside_load_high(i64 %x, i16 %y) {
+; CHECK-LABEL: define i32 @store_inside_load_high(
+; CHECK-SAME: i64 [[X:%.*]], i16 [[Y:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT:%.*]] = zext i16 [[Y]] to i32
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw i32 [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT]], 16
+; CHECK-NEXT: [[TMP0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = and i32 [[TMP0]], 65535
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i32 [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT: ret i32 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+ %alloca = alloca [8 x i8], align 8
+ %alloca.2 = getelementptr inbounds i8, ptr %alloca, i64 2
+ store i64 %x, ptr %alloca
+ store i16 %y, ptr %alloca.2
+ %load = load i32, ptr %alloca
+ ret i32 %load
+}
+
+; Having a larger alloca shouldn't change things.
+
+define i32 @store_inside_load_low_large_alloca(i64 %x, i16 %y) {
+; CHECK-LABEL: define i32 @store_inside_load_low_large_alloca(
+; CHECK-SAME: i64 [[X:%.*]], i16 [[Y:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT:%.*]] = and i32 [[TMP0]], -65536
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = zext i16 [[Y]] to i32
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i32 [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT: ret i32 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+ %alloca = alloca [2048 x i8], align 8
+ store i64 %x, ptr %alloca
+ store i16 %y, ptr %alloca
+ %load = load i32, ptr %alloca
+ ret i32 %load
+}
+
+define i32 @store_inside_load_high_large_alloca(i64 %x, i16 %y) {
+; CHECK-LABEL: define i32 @store_inside_load_high_large_alloca(
+; CHECK-SAME: i64 [[X:%.*]], i16 [[Y:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT:%.*]] = zext i16 [[Y]] to i32
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw i32 [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT]], 16
+; CHECK-NEXT: [[TMP0:%.*]] = trunc i64 [[X]] to i32
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = and i32 [[TMP0]], 65535
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i32 [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT: ret i32 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+ %alloca = alloca [2048 x i8], align 8
+ %alloca.2 = getelementptr inbounds i8, ptr %alloca, i64 2
+ store i64 %x, ptr %alloca
+ store i16 %y, ptr %alloca.2
+ %load = load i32, ptr %alloca
+ ret i32 %load
+}
+
+; Check that we get similar behavior using larger types.
+
+define i512 @store_inside_load_low_large_alloca_large_types(i1024 %x, i256 %y) {
+; CHECK-LABEL: define i512 @store_inside_load_low_large_alloca_large_types(
+; CHECK-SAME: i1024 [[X:%.*]], i256 [[Y:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[TMP0:%.*]] = trunc i1024 [[X]] to i512
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT:%.*]] = and i512 [[TMP0]], -115792089237316195423570985008687907853269984665640564039457584007913129639936
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = zext i256 [[Y]] to i512
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i512 [[ALLOCA_SROA_0_SROA_3_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT: ret i512 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+ %alloca = alloca [2048 x i8], align 8
+ store i1024 %x, ptr %alloca
+ store i256 %y, ptr %alloca
+ %load = load i512, ptr %alloca
+ ret i512 %load
+}
+
+define i512 @store_inside_load_high_large_alloca_large_types(i1024 %x, i256 %y) {
+; CHECK-LABEL: define i512 @store_inside_load_high_large_alloca_large_types(
+; CHECK-SAME: i1024 [[X:%.*]], i256 [[Y:%.*]]) {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT:%.*]] = zext i256 [[Y]] to i512
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT:%.*]] = shl nuw i512 [[ALLOCA_SROA_0_SROA_2_0_INSERT_EXT]], 256
+; CHECK-NEXT: [[TMP0:%.*]] = trunc i1024 [[X]] to i512
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT:%.*]] = and i512 [[TMP0]], 115792089237316195423570985008687907853269984665640564039457584007913129639935
+; CHECK-NEXT: [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT:%.*]] = or disjoint i512 [[ALLOCA_SROA_0_SROA_2_0_INSERT_SHIFT]], [[ALLOCA_SROA_0_SROA_0_0_INSERT_EXT]]
+; CHECK-NEXT: ret i512 [[ALLOCA_SROA_0_SROA_0_0_INSERT_INSERT]]
+;
+entry:
+ %alloca = alloca [2048 x i8], align 8
+ %alloca.32 = getelementptr inbounds i8, ptr %alloca, i64 32
+ store i1024 %x, ptr %alloca
+ store i256 %y, ptr %alloca.32
+ %load = load i512, ptr %alloca
+ ret i512 %load
+}
More information about the llvm-commits
mailing list