[llvm] [AMDGPU] Fix maximum physical LDS reporting for gfx6 (PR #223185)
Chinmay Deshpande via llvm-commits
llvm-commits at lists.llvm.org
Sat Sep 12 15:23:50 PDT 2026
https://github.com/chinmaydd created https://github.com/llvm/llvm-project/pull/223185
Reference: https://docs.amd.com/api/khub/documents/VeItSxvoJNzy28rl~JwEAQ/content#page=77
> “Each compute unit has a 64 kB of Local Data Share (LDS) memory space…”
> “Each workgroup can allocate up to 32 kB of this space, and can read and write any portion of the LDS space allocated to it.”
>From 9f58bae2dda197ac0cf2f36461e895eb453f2bf5 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sat, 12 Sep 2026 18:21:29 -0400
Subject: [PATCH] [AMDGPU] Fix LDS reporting for gfx6
Change-Id: Idaa597f95d7a3028e3848726948892b1f66a220d
---
llvm/lib/Target/AMDGPU/AMDGPU.td | 9 +++++----
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 7 ++++---
llvm/test/CodeGen/AMDGPU/occupancy-levels.ll | 14 +++++++-------
.../unittests/Target/AMDGPU/AMDGPUUnitTests.cpp | 17 +++++++++++++++++
.../unittests/TargetParser/TargetParserTest.cpp | 6 +++++-
5 files changed, 38 insertions(+), 15 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index 30db12b565082e..5eaa5df801690a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -270,9 +270,9 @@ defm LDSMisalignedBug : AMDGPUSubtargetFeature<"lds-misaligned-bug",
/*GenPredicate=*/0
>;
-// Set for gfx10/11/12 where a work-group can address only half of the physical
-// LDS block. So the physical block is twice the addressable size, for example
-// 128k physical and 64k addressable.
+// Set where a work-group can address only half of the physical LDS block:
+// gfx6 has 64 KiB physical and 32 KiB addressable, while gfx10/11/12 have
+// 128 KiB physical and 64 KiB addressable.
defm HalfAddressablePhysicalLocalMemory : AMDGPUSubtargetFeature<"half-addressable-physical-local-memory",
"A work-group can address only half of the physical LDS block",
/*GenPredicate=*/0
@@ -1619,7 +1619,8 @@ class GCNSubtargetFeatureGeneration <string Value,
def FeatureSouthernIslands : GCNSubtargetFeatureGeneration<"SOUTHERN_ISLANDS",
"southern-islands",
- [FeatureFP64, FeatureAddressableLocalMemorySize32768, FeatureMIMG_R128,
+ [FeatureFP64, FeatureAddressableLocalMemorySize32768,
+ FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128,
FeatureWavefrontSize64, FeatureSupportsWave64, FeatureSMemTimeInst,
FeatureMadMacF32Insts,
FeatureDsSrc2Insts, FeatureLDSBankCount32, FeatureMovrel,
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 88e2d2f485262f..58ca7e69cecf80 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1130,8 +1130,9 @@ static unsigned getMaxHWAddressableLocalMemorySize(const MCSubtargetInfo &STI) {
// Total physical size of LDS on the block, in bytes. On targets with
// FeatureHalfAddressablePhysicalLocalMemory the physical block is twice the
-// addressable size (gfx10/11/12, 128k physical and 64k addressable). On other
-// targets it is equal to the addressable size.
+// addressable size (gfx6: 64 KiB physical and 32 KiB addressable;
+// gfx10/11/12: 128 KiB physical and 64 KiB addressable). On other targets it is
+// equal to the addressable size.
static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI) {
unsigned Addressable = getMaxHWAddressableLocalMemorySize(STI);
if (STI.getFeatureBits().test(FeatureHalfAddressablePhysicalLocalMemory))
@@ -1140,7 +1141,7 @@ static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI) {
}
// Sizes in use, by generation (addressable / physical block):
-// gfx6 : 32 KiB
+// gfx6 : 32 KiB addressable, 64 KiB physical block
// gfx7 / gfx8 / gfx9: 64 KiB
// gfx9.5 (gfx950) : 160 KiB
// gfx10 / 11 / 12 : 64 KiB addressable, 128 KiB physical block
diff --git a/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll b/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll
index b7ee93228e7aae..3299f9d1ce2f63 100644
--- a/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll
+++ b/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll
@@ -656,7 +656,7 @@ define amdgpu_kernel void @used_lds_13112() {
}
; GCN-LABEL: {{^}}used_lds_8252_max_group_size_64:
-; GFX6: ; Occupancy: 1{{$}}
+; GFX6: ; Occupancy: 2{{$}}
; GFX7: ; Occupancy: 2{{$}}
; GFX8: ; Occupancy: 2{{$}}
; GFX9: ; Occupancy: 2{{$}}
@@ -680,7 +680,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_64() #3 {
}
; GCN-LABEL: {{^}}used_lds_8252_max_group_size_96:
-; GFX6: ; Occupancy: 2{{$}}
+; GFX6: ; Occupancy: 4{{$}}
; GFX7: ; Occupancy: 4{{$}}
; GFX8: ; Occupancy: 4{{$}}
; GFX9: ; Occupancy: 4{{$}}
@@ -704,7 +704,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_96() #4 {
}
; GCN-LABEL: {{^}}used_lds_8252_max_group_size_128:
-; GFX6: ; Occupancy: 2{{$}}
+; GFX6: ; Occupancy: 4{{$}}
; GFX7: ; Occupancy: 4{{$}}
; GFX8: ; Occupancy: 4{{$}}
; GFX9: ; Occupancy: 4{{$}}
@@ -727,7 +727,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_128() #5 {
}
; GCN-LABEL: {{^}}used_lds_8252_max_group_size_192:
-; GFX6: ; Occupancy: 3{{$}}
+; GFX6: ; Occupancy: 6{{$}}
; GFX7: ; Occupancy: 6{{$}}
; GFX8: ; Occupancy: 6{{$}}
; GFX9: ; Occupancy: 6{{$}}
@@ -750,7 +750,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_192() #6 {
}
; GCN-LABEL: {{^}}used_lds_8252_max_group_size_256:
-; GFX6: ; Occupancy: 3{{$}}
+; GFX6: ; Occupancy: 7{{$}}
; GFX7: ; Occupancy: 7{{$}}
; GFX8: ; Occupancy: 7{{$}}
; GFX9: ; Occupancy: 7{{$}}
@@ -771,7 +771,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_256() #7 {
}
; GCN-LABEL: {{^}}used_lds_8252_max_group_size_512:
-; GFX6: ; Occupancy: 6{{$}}
+; GFX6: ; Occupancy: 10{{$}}
; GFX7: ; Occupancy: 10{{$}}
; GFX8: ; Occupancy: 10{{$}}
; GFX9: ; Occupancy: 10{{$}}
@@ -806,7 +806,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_1024() #9 {
}
; GCN-LABEL: {{^}}used_lds_8252_max_group_size_32:
-; GFX6: ; Occupancy: 1{{$}}
+; GFX6: ; Occupancy: 2{{$}}
; GFX7: ; Occupancy: 2{{$}}
; GFX8: ; Occupancy: 2{{$}}
; GFX9: ; Occupancy: 2{{$}}
diff --git a/llvm/unittests/Target/AMDGPU/AMDGPUUnitTests.cpp b/llvm/unittests/Target/AMDGPU/AMDGPUUnitTests.cpp
index 51fda88e66ecac..33b4eca0a1dd30 100644
--- a/llvm/unittests/Target/AMDGPU/AMDGPUUnitTests.cpp
+++ b/llvm/unittests/Target/AMDGPU/AMDGPUUnitTests.cpp
@@ -276,6 +276,23 @@ TEST_F(AMDGPUTestBase, TestOccupancyAbsoluteLimits) {
256);
}
+TEST_F(AMDGPUTestBase, TestGFX6LocalMemorySize) {
+ for (StringRef CPU : {"gfx600", "gfx601", "gfx602"}) {
+ SCOPED_TRACE(CPU.str());
+ auto TM = createAMDGPUTargetMachine(Triple("amdgcn-amd-amdhsa"), CPU, "");
+ ASSERT_NE(TM, nullptr);
+ GCNSubtarget ST(TM->getTargetTriple(), CPU.str(), "", *TM);
+
+ EXPECT_EQ(ST.getLocalMemorySize(), 65536u);
+ EXPECT_EQ(ST.getAddressableLocalMemorySize(), 32768u);
+
+ // Two workgroups using 32 KiB each fit in a CU. With 256 threads per
+ // workgroup, each SIMD can therefore hold two waves, limited only by LDS.
+ EXPECT_EQ(ST.getOccupancyWithWorkGroupSizes(32768, {256, 256}),
+ std::make_pair(2u, 2u));
+ }
+}
+
static const char *printSubReg(const TargetRegisterInfo &TRI, unsigned SubReg) {
return SubReg ? TRI.getSubRegIndexName(SubReg) : "<none>";
}
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 0037bc63e7b0e2..550185ebe820de 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -2867,7 +2867,11 @@ TEST(TargetParserTest, testAMDGPUHalfAddressableLDSFeature) {
AMDGPU::FEAT_HALF_ADDRESSABLE_PHYSICAL_LOCAL_MEMORY);
};
- // Only gfx10/11/12 address half of the physical LDS block.
+ // Gfx6 and gfx10/11/12 address half of the physical LDS block.
+ EXPECT_TRUE(Has(AMDGPU::GK_GFX600));
+ EXPECT_TRUE(Has(AMDGPU::GK_GFX601));
+ EXPECT_TRUE(Has(AMDGPU::GK_GFX602));
+ EXPECT_FALSE(Has(AMDGPU::GK_GFX700));
EXPECT_FALSE(Has(AMDGPU::GK_GFX900));
EXPECT_TRUE(Has(AMDGPU::GK_GFX1030));
EXPECT_TRUE(Has(AMDGPU::GK_GFX1100));
More information about the llvm-commits
mailing list