[llvm] [Attributor] Don't heap-to-stack an allocation whose size is unknown (PR #221867)
Larry Meadows via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 8 13:56:22 PDT 2026
https://github.com/lfmeadow updated https://github.com/llvm/llvm-project/pull/221867
>From d0585a501ffb6959168cf59c030dccfbd63f7602 Mon Sep 17 00:00:00 2001
From: Larry Meadows <Lawrence.Meadows at amd.com>
Date: Tue, 8 Sep 2026 15:54:33 -0500
Subject: [PATCH] [Attributor] Don't heap-to-stack an allocation whose size is
unknown
AAHeapToStackFunction::manifest() asserts that ObjectSizeOffsetEvaluator can
produce a size for the allocation it is converting, but nothing establishes
that beforehand. An allocation function is recognized by its allockind
attribute, which does not imply allocsize, so a call whose size the evaluator
cannot compute reaches manifest() and crashes.
Such an allocation is normally rejected by accident: updateImpl() invalidates
it because getSize() returned nothing and the -max-heap-to-stack-size cap
comparison therefore fails. Globalized locals are exempt from that cap, since
moving one to the stack is worthwhile however large it is, and the exemption
takes the unknown-size check with it. Passing -max-heap-to-stack-size=-1
reaches the same assert for any allocator.
Check for a computable size separately from the cap, using a new
hasComputableAllocSize() that mirrors what the evaluator accepts. The check
belongs in updateImpl() rather than manifest(): AAKernelInfo::updateImpl()
consults isAssumedHeapToStack() when deciding that a store needs no SPMD
guard, so declining the conversion after the fixpoint would leave that store
unguarded.
One way to produce the sizeless shape is DeadArgumentElimination, which drops
allocsize when it deletes an argument the attribute refers to -- it has to,
since the verifier rejects an out-of-range allocsize index -- while leaving
allockind and alloc-family in place.
---
llvm/include/llvm/Analysis/MemoryBuiltins.h | 6 +++
llvm/lib/Analysis/MemoryBuiltins.cpp | 7 +++
.../Transforms/IPO/AttributorAttributes.cpp | 12 +++++
.../Attributor/heap_to_stack_gpu.ll | 47 ++++++++++++++-----
4 files changed, 60 insertions(+), 12 deletions(-)
diff --git a/llvm/include/llvm/Analysis/MemoryBuiltins.h b/llvm/include/llvm/Analysis/MemoryBuiltins.h
index ebe644cee53e1..e964976c59f03 100644
--- a/llvm/include/llvm/Analysis/MemoryBuiltins.h
+++ b/llvm/include/llvm/Analysis/MemoryBuiltins.h
@@ -113,6 +113,12 @@ LLVM_ABI std::optional<APInt> getAllocSize(
return V;
});
+/// Return true if the size of the allocation performed by \p CB can be
+/// determined, as a constant or as a value ObjectSizeOffsetEvaluator can
+/// materialize at runtime.
+LLVM_ABI bool hasComputableAllocSize(const CallBase *CB,
+ const TargetLibraryInfo *TLI);
+
/// If this is a call to an allocation function that initializes memory to a
/// fixed value, return said value in the requested type. Otherwise, return
/// nullptr.
diff --git a/llvm/lib/Analysis/MemoryBuiltins.cpp b/llvm/lib/Analysis/MemoryBuiltins.cpp
index e208b3d9317b6..3f5b2d1654205 100644
--- a/llvm/lib/Analysis/MemoryBuiltins.cpp
+++ b/llvm/lib/Analysis/MemoryBuiltins.cpp
@@ -411,6 +411,13 @@ llvm::getAllocSize(const CallBase *CB, const TargetLibraryInfo *TLI,
return Size;
}
+bool llvm::hasComputableAllocSize(const CallBase *CB,
+ const TargetLibraryInfo *TLI) {
+ // Keep in sync with ObjectSizeOffsetEvaluator::visitCallBase.
+ std::optional<AllocFnsTy> FnData = getAllocationSize(CB, TLI);
+ return FnData && FnData->AllocTy != StrDupLike;
+}
+
Constant *llvm::getInitialValueOfAllocation(const Value *V,
const TargetLibraryInfo *TLI,
Type *Ty) {
diff --git a/llvm/lib/Transforms/IPO/AttributorAttributes.cpp b/llvm/lib/Transforms/IPO/AttributorAttributes.cpp
index 22af6f7741544..d9e0d91fba911 100644
--- a/llvm/lib/Transforms/IPO/AttributorAttributes.cpp
+++ b/llvm/lib/Transforms/IPO/AttributorAttributes.cpp
@@ -7282,6 +7282,18 @@ ChangeStatus AAHeapToStackFunction::updateImpl(Attributor &A) {
}
std::optional<APInt> Size = getSize(A, *this, AI);
+
+ // manifest() needs a size, either the constant above or one
+ // ObjectSizeOffsetEvaluator can materialize.
+ if (!Size && !hasComputableAllocSize(AI.CB, TLI)) {
+ LLVM_DEBUG(dbgs() << "[H2S] Unsizable allocation: " << *AI.CB << "\n");
+ AI.Status = AllocationInfo::INVALID;
+ Changed = ChangeStatus::CHANGED;
+ continue;
+ }
+
+ // A globalized local is exempt from the size cap: moving it to the stack is
+ // worthwhile however large it is.
if (!AI.IsGlobalizedLocal && MaxHeapToStackSize != -1) {
if (!Size || Size->ugt(MaxHeapToStackSize)) {
LLVM_DEBUG({
diff --git a/llvm/test/Transforms/Attributor/heap_to_stack_gpu.ll b/llvm/test/Transforms/Attributor/heap_to_stack_gpu.ll
index c85e496c497f6..1f6f6e34ddfc3 100644
--- a/llvm/test/Transforms/Attributor/heap_to_stack_gpu.ll
+++ b/llvm/test/Transforms/Attributor/heap_to_stack_gpu.ll
@@ -304,7 +304,7 @@ define void @test9() {
; CHECK-NEXT: [[I:%.*]] = tail call noalias ptr @malloc(i64 noundef 4)
; CHECK-NEXT: tail call void @no_sync_func(ptr nofree captures(none) [[I]])
; CHECK-NEXT: store i32 10, ptr [[I]], align 4
-; CHECK-NEXT: tail call void @foo_nounw(ptr nofree nonnull align 4 dereferenceable(4) [[I]]) #[[ATTR8:[0-9]+]]
+; CHECK-NEXT: tail call void @foo_nounw(ptr nofree nonnull align 4 dereferenceable(4) [[I]]) #[[ATTR9:[0-9]+]]
; CHECK-NEXT: tail call void @free(ptr nonnull align 4 captures(none) dereferenceable(4) [[I]])
; CHECK-NEXT: ret void
;
@@ -345,7 +345,7 @@ define void @test11() {
; CHECK-LABEL: define {{[^@]+}}@test11() {
; CHECK-NEXT: bb:
; CHECK-NEXT: [[I:%.*]] = tail call noalias ptr @malloc(i64 noundef 4)
-; CHECK-NEXT: tail call void @sync_will_return(ptr [[I]]) #[[ATTR8]]
+; CHECK-NEXT: tail call void @sync_will_return(ptr [[I]]) #[[ATTR9]]
; CHECK-NEXT: tail call void @free(ptr captures(none) [[I]])
; CHECK-NEXT: ret void
;
@@ -599,7 +599,7 @@ define void @test16c(i8 %v, ptr %P) {
; CHECK-NEXT: bb:
; CHECK-NEXT: [[I:%.*]] = tail call noalias ptr @malloc(i64 noundef 4)
; CHECK-NEXT: store ptr [[I]], ptr [[P]], align 8
-; CHECK-NEXT: tail call void @no_sync_func(ptr nofree captures(none) [[I]]) #[[ATTR8]]
+; CHECK-NEXT: tail call void @no_sync_func(ptr nofree captures(none) [[I]]) #[[ATTR9]]
; CHECK-NEXT: tail call void @free(ptr captures(none) [[I]])
; CHECK-NEXT: ret void
;
@@ -633,7 +633,7 @@ define void @test17() {
; CHECK-NEXT: bb:
; CHECK-NEXT: [[I_H2S:%.*]] = alloca i8, i64 4, align 1, addrspace(5)
; CHECK-NEXT: [[MALLOC_CAST:%.*]] = addrspacecast ptr addrspace(5) [[I_H2S]] to ptr
-; CHECK-NEXT: tail call void @usei8(ptr noalias nofree captures(none) [[MALLOC_CAST]]) #[[ATTR9:[0-9]+]]
+; CHECK-NEXT: tail call void @usei8(ptr noalias nofree captures(none) [[MALLOC_CAST]]) #[[ATTR10:[0-9]+]]
; CHECK-NEXT: ret void
;
bb:
@@ -647,7 +647,7 @@ define void @test17b() {
; CHECK-LABEL: define {{[^@]+}}@test17b() {
; CHECK-NEXT: bb:
; CHECK-NEXT: [[I:%.*]] = tail call noalias ptr @__kmpc_alloc_shared(i64 noundef 4)
-; CHECK-NEXT: tail call void @usei8(ptr nofree [[I]]) #[[ATTR9]]
+; CHECK-NEXT: tail call void @usei8(ptr nofree [[I]]) #[[ATTR10]]
; CHECK-NEXT: tail call void @__kmpc_free_shared(ptr captures(none) [[I]], i64 noundef 4)
; CHECK-NEXT: ret void
;
@@ -665,7 +665,7 @@ define void @move_alloca() {
; CHECK-NEXT: br label [[NOT_ENTRY:%.*]]
; CHECK: not_entry:
; CHECK-NEXT: [[MALLOC_CAST:%.*]] = addrspacecast ptr addrspace(5) [[I_H2S]] to ptr
-; CHECK-NEXT: tail call void @usei8(ptr noalias nofree captures(none) [[MALLOC_CAST]]) #[[ATTR9]]
+; CHECK-NEXT: tail call void @usei8(ptr noalias nofree captures(none) [[MALLOC_CAST]]) #[[ATTR10]]
; CHECK-NEXT: ret void
;
entry:
@@ -686,7 +686,7 @@ define void @test16e(i8 %v) norecurse {
; CHECK-NEXT: bb:
; CHECK-NEXT: [[I:%.*]] = tail call noalias ptr @__kmpc_alloc_shared(i64 noundef 4)
; CHECK-NEXT: store ptr [[I]], ptr @G, align 8
-; CHECK-NEXT: call void @usei8(ptr nofree captures(none) [[I]]) #[[ATTR10:[0-9]+]]
+; CHECK-NEXT: call void @usei8(ptr nofree captures(none) [[I]]) #[[ATTR11:[0-9]+]]
; CHECK-NEXT: tail call void @__kmpc_free_shared(ptr noalias captures(none) [[I]], i64 noundef 4)
; CHECK-NEXT: ret void
;
@@ -708,7 +708,7 @@ define void @test16f(i8 %v) norecurse {
; CHECK-NEXT: [[I_H2S:%.*]] = alloca i8, i64 4, align 1, addrspace(5)
; CHECK-NEXT: [[MALLOC_CAST:%.*]] = addrspacecast ptr addrspace(5) [[I_H2S]] to ptr
; CHECK-NEXT: store ptr [[MALLOC_CAST]], ptr @Gtl, align 8
-; CHECK-NEXT: call void @usei8(ptr nofree captures(none) [[MALLOC_CAST]]) #[[ATTR10]]
+; CHECK-NEXT: call void @usei8(ptr nofree captures(none) [[MALLOC_CAST]]) #[[ATTR11]]
; CHECK-NEXT: ret void
;
bb:
@@ -725,7 +725,7 @@ define void @convert_large_kmpc_alloc_shared() {
; CHECK-NEXT: bb:
; CHECK-NEXT: [[I_H2S:%.*]] = alloca i8, i64 256, align 1, addrspace(5)
; CHECK-NEXT: [[MALLOC_CAST:%.*]] = addrspacecast ptr addrspace(5) [[I_H2S]] to ptr
-; CHECK-NEXT: tail call void @usei8(ptr noalias nofree captures(none) [[MALLOC_CAST]]) #[[ATTR9]]
+; CHECK-NEXT: tail call void @usei8(ptr noalias nofree captures(none) [[MALLOC_CAST]]) #[[ATTR10]]
; CHECK-NEXT: ret void
;
bb:
@@ -735,6 +735,28 @@ bb:
ret void
}
+; A globalized-local allocator with no allocsize, which is what
+; DeadArgumentElimination leaves behind once it deletes a dead size argument.
+; The size is not recoverable, so the allocation has to stay on the heap even
+; though globalized locals are exempt from the size cap.
+
+declare ptr @__kmpc_alloc_shared_no_size() allockind("alloc,uninitialized") "alloc-family"="__kmpc_alloc_shared"
+
+define void @no_allocsize_kmpc_alloc_shared() {
+; CHECK-LABEL: define {{[^@]+}}@no_allocsize_kmpc_alloc_shared() {
+; CHECK-NEXT: bb:
+; CHECK-NEXT: [[I:%.*]] = tail call noalias ptr @__kmpc_alloc_shared_no_size()
+; CHECK-NEXT: tail call void @usei8(ptr noalias nofree captures(none) [[I]]) #[[ATTR10]]
+; CHECK-NEXT: tail call void @__kmpc_free_shared(ptr noalias captures(none) [[I]], i64 noundef 4)
+; CHECK-NEXT: ret void
+;
+bb:
+ %i = tail call noalias ptr @__kmpc_alloc_shared_no_size()
+ tail call void @usei8(ptr nocapture nofree %i) nosync nounwind willreturn
+ tail call void @__kmpc_free_shared(ptr %i, i64 4)
+ ret void
+}
+
;.
; CHECK: attributes #[[ATTR0:[0-9]+]] = { nounwind willreturn }
@@ -745,9 +767,10 @@ bb:
; CHECK: attributes #[[ATTR5:[0-9]+]] = { allockind("alloc,uninitialized") allocsize(0) "alloc-family"="__kmpc_alloc_shared" }
; CHECK: attributes #[[ATTR6:[0-9]+]] = { allockind("free") "alloc-family"="__kmpc_alloc_shared" }
; CHECK: attributes #[[ATTR7]] = { norecurse }
-; CHECK: attributes #[[ATTR8]] = { nounwind }
-; CHECK: attributes #[[ATTR9]] = { nosync nounwind willreturn }
-; CHECK: attributes #[[ATTR10]] = { nocallback nosync nounwind willreturn }
+; CHECK: attributes #[[ATTR8:[0-9]+]] = { allockind("alloc,uninitialized") "alloc-family"="__kmpc_alloc_shared" }
+; CHECK: attributes #[[ATTR9]] = { nounwind }
+; CHECK: attributes #[[ATTR10]] = { nosync nounwind willreturn }
+; CHECK: attributes #[[ATTR11]] = { nocallback nosync nounwind willreturn }
;.
;; NOTE: These prefixes are unused and the list is autogenerated. Do not add tests below this line:
; CGSCC: {{.*}}
More information about the llvm-commits
mailing list