[llvm] [AMDGPU] Align to LDS granularity in getMaxLocalMemSizeWithWaveCount (PR #219172)
Frederik Harwath via llvm-commits
llvm-commits at lists.llvm.org
Thu Aug 27 03:56:40 PDT 2026
https://github.com/frederik-h updated https://github.com/llvm/llvm-project/pull/219172
>From 6ef4e1f3b6fd95bba6ea6414c5c1c550f3e97ae2 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Thu, 27 Aug 2026 04:21:09 -0400
Subject: [PATCH 1/3] [AMDGPU] Add test documenting alloc promotion bug
---
.../AMDGPU/large-work-group-promote-alloca.ll | 2 +-
.../promote-alloca-lds-granularity-limit.ll | 36 +++++++++++++++++++
2 files changed, 37 insertions(+), 1 deletion(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll
diff --git a/llvm/test/CodeGen/AMDGPU/large-work-group-promote-alloca.ll b/llvm/test/CodeGen/AMDGPU/large-work-group-promote-alloca.ll
index 32b356a514cbd..89f43511c8da5 100644
--- a/llvm/test/CodeGen/AMDGPU/large-work-group-promote-alloca.ll
+++ b/llvm/test/CodeGen/AMDGPU/large-work-group-promote-alloca.ll
@@ -216,7 +216,7 @@ entry:
; SI-LABEL: @occupancy_9(
; CI-LABEL: @occupancy_9(
; SI: alloca
-; CI-NOT: alloca
+; CI: alloca [28 x i8]
define amdgpu_kernel void @occupancy_9(ptr addrspace(1) nocapture %out, ptr addrspace(1) nocapture %in) #7 {
entry:
%stack = alloca [28 x i8], align 4, addrspace(5)
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll
new file mode 100644
index 0000000000000..e35a322d0f6ca
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll
@@ -0,0 +1,36 @@
+; RUN: opt -S -mtriple=amdgpu9.50-amd-amdhsa -passes=amdgpu-promote-alloca -disable-promote-alloca-to-vector < %s | FileCheck %s
+
+ at lds_12800 = internal addrspace(3) global [12800 x i8] poison, align 16
+attributes #0 = { "amdgpu-flat-work-group-size"="64,64" }
+
+; This is a regression test for a bug in getMaxLocalMemSizeWithWaveCount
+; which does not round to the LDS allocation block size, leading
+; AMDGPUPromoteAlloca pass to overestimate the available LDS.
+
+; We have LocalMemorySize = 163840, allowing for floor(163840/ 12800)
+; = 12 workgroups and occupancy 12 / 4 = 3 with the 12800 bytes of LDS
+; usage and one wave per workgroup.
+
+; Without aligning down to the LDS granularity of 1280 in
+; getMaxLocalMemSizeWithWaveCount, the limit in alloca promotion is
+; 163840 / 12 = 13653 bytes which leads to promotion of the alloca
+; in the function.
+
+; With rounding down, the promotion limit is 12800 bytes. The alloca
+; would add 64 * 10 bytes which exceeds the promotion limit.
+
+; FIXME The alloca should not get promoted
+; CHECK-NOT: alloca
+
+define amdgpu_kernel void @test(ptr addrspace(1) %out, i32 %idx) #0 {
+
+ %stack = alloca [10 x i8], align 1, addrspace(5)
+ %lds.ptr = getelementptr inbounds [12800 x i8], ptr addrspace(3) @lds_12800, i32 0, i32 0
+ store volatile i8 1, ptr addrspace(3) %lds.ptr, align 1
+
+ %arrayidx = getelementptr inbounds [10 x i8], ptr addrspace(5) %stack, i32 0, i32 %idx
+ store i8 7, ptr addrspace(5) %arrayidx, align 1
+ %load = load i8, ptr addrspace(5) %arrayidx, align 1
+ store i8 %load, ptr addrspace(1) %out, align 1
+ ret void
+}
\ No newline at end of file
>From cca3f834c82175cce09e49e6e35a12795d59ce14 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <fharwath at amd.com>
Date: Thu, 27 Aug 2026 04:26:25 -0400
Subject: [PATCH 2/3] [AMDGPU] Align to LDS granularity in
getMaxLocalMemSizeWithWaveCount
The AMDGPUSutarget was recently extended by a LDSAllocationGranularity
field. This was added to make the occupancy calculation in
getOccupancyWithWorkgroupSize more precise by rounding the LDS
allocation size to the allocation granularity. The
getMaxLocalMemSizeWithWaveCount does an "inverse" computation,
computing available LDS from occupancy. This should also round to a
multiple of the LDS allocation granularity since otherwise the
available LDS may be overestimated. This leads to incorrect alloca
promotions in the AMDGPUPromoteAlloca pass.
Add rounding to getMaxLocalMemSizeWithWaveCount and add a test
demonstrating the impact on AMDGPUPromoteAlloca.
---
llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp | 3 ++-
.../CodeGen/AMDGPU/large-work-group-promote-alloca.ll | 2 +-
.../AMDGPU/promote-alloca-lds-granularity-limit.ll | 8 ++++----
3 files changed, 7 insertions(+), 6 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
index 87515d22ed422..eb90a658e06aa 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
@@ -46,7 +46,8 @@ AMDGPUSubtarget::getMaxLocalMemSizeWithWaveCount(unsigned NWaves,
const unsigned WorkGroupsPerCU =
std::max(1u, (NWaves * getEUsPerCU()) / WavesPerWorkgroup);
- return getLocalMemorySize() / WorkGroupsPerCU;
+ const unsigned Granularity = std::max(LDSAllocationGranularity, 1u);
+ return alignDown(getLocalMemorySize() / WorkGroupsPerCU, Granularity);
}
std::pair<unsigned, unsigned> AMDGPUSubtarget::getOccupancyWithWorkGroupSizes(
diff --git a/llvm/test/CodeGen/AMDGPU/large-work-group-promote-alloca.ll b/llvm/test/CodeGen/AMDGPU/large-work-group-promote-alloca.ll
index 89f43511c8da5..dc688121b4130 100644
--- a/llvm/test/CodeGen/AMDGPU/large-work-group-promote-alloca.ll
+++ b/llvm/test/CodeGen/AMDGPU/large-work-group-promote-alloca.ll
@@ -116,7 +116,7 @@ entry:
; SI-LABEL: @occupancy_6(
; CI-LABEL: @occupancy_6(
; SI: alloca
-; CI-NOT: alloca
+; CI: alloca [42 x i8]
define amdgpu_kernel void @occupancy_6(ptr addrspace(1) nocapture %out, ptr addrspace(1) nocapture %in) #5 {
entry:
%stack = alloca [42 x i8], align 4, addrspace(5)
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll
index e35a322d0f6ca..79bdca36a8315 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll
@@ -4,7 +4,7 @@
attributes #0 = { "amdgpu-flat-work-group-size"="64,64" }
; This is a regression test for a bug in getMaxLocalMemSizeWithWaveCount
-; which does not round to the LDS allocation block size, leading
+; which did not round to the LDS allocation block size, leading
; AMDGPUPromoteAlloca pass to overestimate the available LDS.
; We have LocalMemorySize = 163840, allowing for floor(163840/ 12800)
@@ -13,14 +13,14 @@ attributes #0 = { "amdgpu-flat-work-group-size"="64,64" }
; Without aligning down to the LDS granularity of 1280 in
; getMaxLocalMemSizeWithWaveCount, the limit in alloca promotion is
-; 163840 / 12 = 13653 bytes which leads to promotion of the alloca
+; 163840 / 12 = 13653 bytes which led to the promotion of the alloca
; in the function.
; With rounding down, the promotion limit is 12800 bytes. The alloca
; would add 64 * 10 bytes which exceeds the promotion limit.
+; This prevents the promotion of the alloca.
-; FIXME The alloca should not get promoted
-; CHECK-NOT: alloca
+; CHECK: alloca
define amdgpu_kernel void @test(ptr addrspace(1) %out, i32 %idx) #0 {
>From f876afee61076ceea3e65f41508ab92c7b434885 Mon Sep 17 00:00:00 2001
From: Frederik Harwath <frederik at harwath.name>
Date: Thu, 27 Aug 2026 12:56:30 +0200
Subject: [PATCH 3/3] copilot review change
Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot at users.noreply.github.com>
---
.../CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll | 3 ++-
1 file changed, 2 insertions(+), 1 deletion(-)
diff --git a/llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll b/llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll
index 79bdca36a8315..98741f40e0366 100644
--- a/llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll
+++ b/llvm/test/CodeGen/AMDGPU/promote-alloca-lds-granularity-limit.ll
@@ -20,7 +20,8 @@ attributes #0 = { "amdgpu-flat-work-group-size"="64,64" }
; would add 64 * 10 bytes which exceeds the promotion limit.
; This prevents the promotion of the alloca.
-; CHECK: alloca
+; CHECK-LABEL: @test(
+; CHECK: %stack = alloca [10 x i8], align 1, addrspace(5)
define amdgpu_kernel void @test(ptr addrspace(1) %out, i32 %idx) #0 {
More information about the llvm-commits
mailing list