[llvm] [AMDGPU] Fix maximum physical LDS reporting for gfx6 (PR #223185)

via llvm-commits llvm-commits at lists.llvm.org
Sat Sep 12 15:24:25 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-amdgpu

Author: Chinmay Deshpande (chinmaydd)

<details>
<summary>Changes</summary>

Reference: https://docs.amd.com/api/khub/documents/VeItSxvoJNzy28rl~JwEAQ/content#page=77

> “Each compute unit has a 64 kB of Local Data Share (LDS) memory space…”
> “Each workgroup can allocate up to 32 kB of this space, and can read and write any portion of the LDS space allocated to it.”

---
Full diff: https://github.com/llvm/llvm-project/pull/223185.diff


5 Files Affected:

- (modified) llvm/lib/Target/AMDGPU/AMDGPU.td (+5-4) 
- (modified) llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp (+4-3) 
- (modified) llvm/test/CodeGen/AMDGPU/occupancy-levels.ll (+7-7) 
- (modified) llvm/unittests/Target/AMDGPU/AMDGPUUnitTests.cpp (+17) 
- (modified) llvm/unittests/TargetParser/TargetParserTest.cpp (+5-1) 


``````````diff
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index 30db12b565082e..5eaa5df801690a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -270,9 +270,9 @@ defm LDSMisalignedBug : AMDGPUSubtargetFeature<"lds-misaligned-bug",
   /*GenPredicate=*/0
 >;
 
-// Set for gfx10/11/12 where a work-group can address only half of the physical
-// LDS block. So the physical block is twice the addressable size, for example
-// 128k physical and 64k addressable.
+// Set where a work-group can address only half of the physical LDS block:
+// gfx6 has 64 KiB physical and 32 KiB addressable, while gfx10/11/12 have
+// 128 KiB physical and 64 KiB addressable.
 defm HalfAddressablePhysicalLocalMemory : AMDGPUSubtargetFeature<"half-addressable-physical-local-memory",
   "A work-group can address only half of the physical LDS block",
   /*GenPredicate=*/0
@@ -1619,7 +1619,8 @@ class GCNSubtargetFeatureGeneration <string Value,
 
 def FeatureSouthernIslands : GCNSubtargetFeatureGeneration<"SOUTHERN_ISLANDS",
     "southern-islands",
-  [FeatureFP64, FeatureAddressableLocalMemorySize32768, FeatureMIMG_R128,
+  [FeatureFP64, FeatureAddressableLocalMemorySize32768,
+  FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128,
   FeatureWavefrontSize64, FeatureSupportsWave64, FeatureSMemTimeInst,
   FeatureMadMacF32Insts,
   FeatureDsSrc2Insts, FeatureLDSBankCount32, FeatureMovrel,
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 88e2d2f485262f..58ca7e69cecf80 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1130,8 +1130,9 @@ static unsigned getMaxHWAddressableLocalMemorySize(const MCSubtargetInfo &STI) {
 
 // Total physical size of LDS on the block, in bytes. On targets with
 // FeatureHalfAddressablePhysicalLocalMemory the physical block is twice the
-// addressable size (gfx10/11/12, 128k physical and 64k addressable). On other
-// targets it is equal to the addressable size.
+// addressable size (gfx6: 64 KiB physical and 32 KiB addressable;
+// gfx10/11/12: 128 KiB physical and 64 KiB addressable). On other targets it is
+// equal to the addressable size.
 static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI) {
   unsigned Addressable = getMaxHWAddressableLocalMemorySize(STI);
   if (STI.getFeatureBits().test(FeatureHalfAddressablePhysicalLocalMemory))
@@ -1140,7 +1141,7 @@ static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI) {
 }
 
 // Sizes in use, by generation (addressable / physical block):
-//   gfx6              :  32 KiB
+//   gfx6              :  32 KiB addressable, 64 KiB physical block
 //   gfx7 / gfx8 / gfx9:  64 KiB
 //   gfx9.5 (gfx950)   : 160 KiB
 //   gfx10 / 11 / 12   :  64 KiB addressable, 128 KiB physical block
diff --git a/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll b/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll
index b7ee93228e7aae..3299f9d1ce2f63 100644
--- a/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll
+++ b/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll
@@ -656,7 +656,7 @@ define amdgpu_kernel void @used_lds_13112() {
 }
 
 ; GCN-LABEL: {{^}}used_lds_8252_max_group_size_64:
-; GFX6:       ; Occupancy: 1{{$}}
+; GFX6:       ; Occupancy: 2{{$}}
 ; GFX7:       ; Occupancy: 2{{$}}
 ; GFX8:       ; Occupancy: 2{{$}}
 ; GFX9:       ; Occupancy: 2{{$}}
@@ -680,7 +680,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_64() #3 {
 }
 
 ; GCN-LABEL: {{^}}used_lds_8252_max_group_size_96:
-; GFX6:       ; Occupancy: 2{{$}}
+; GFX6:       ; Occupancy: 4{{$}}
 ; GFX7:       ; Occupancy: 4{{$}}
 ; GFX8:       ; Occupancy: 4{{$}}
 ; GFX9:       ; Occupancy: 4{{$}}
@@ -704,7 +704,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_96() #4 {
 }
 
 ; GCN-LABEL: {{^}}used_lds_8252_max_group_size_128:
-; GFX6:       ; Occupancy: 2{{$}}
+; GFX6:       ; Occupancy: 4{{$}}
 ; GFX7:       ; Occupancy: 4{{$}}
 ; GFX8:       ; Occupancy: 4{{$}}
 ; GFX9:       ; Occupancy: 4{{$}}
@@ -727,7 +727,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_128() #5 {
 }
 
 ; GCN-LABEL: {{^}}used_lds_8252_max_group_size_192:
-; GFX6:       ; Occupancy: 3{{$}}
+; GFX6:       ; Occupancy: 6{{$}}
 ; GFX7:       ; Occupancy: 6{{$}}
 ; GFX8:       ; Occupancy: 6{{$}}
 ; GFX9:       ; Occupancy: 6{{$}}
@@ -750,7 +750,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_192() #6 {
 }
 
 ; GCN-LABEL: {{^}}used_lds_8252_max_group_size_256:
-; GFX6:       ; Occupancy: 3{{$}}
+; GFX6:       ; Occupancy: 7{{$}}
 ; GFX7:       ; Occupancy: 7{{$}}
 ; GFX8:       ; Occupancy: 7{{$}}
 ; GFX9:       ; Occupancy: 7{{$}}
@@ -771,7 +771,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_256() #7 {
 }
 
 ; GCN-LABEL: {{^}}used_lds_8252_max_group_size_512:
-; GFX6:       ; Occupancy: 6{{$}}
+; GFX6:       ; Occupancy: 10{{$}}
 ; GFX7:       ; Occupancy: 10{{$}}
 ; GFX8:       ; Occupancy: 10{{$}}
 ; GFX9:       ; Occupancy: 10{{$}}
@@ -806,7 +806,7 @@ define amdgpu_kernel void @used_lds_8252_max_group_size_1024() #9 {
 }
 
 ; GCN-LABEL: {{^}}used_lds_8252_max_group_size_32:
-; GFX6:       ; Occupancy: 1{{$}}
+; GFX6:       ; Occupancy: 2{{$}}
 ; GFX7:       ; Occupancy: 2{{$}}
 ; GFX8:       ; Occupancy: 2{{$}}
 ; GFX9:       ; Occupancy: 2{{$}}
diff --git a/llvm/unittests/Target/AMDGPU/AMDGPUUnitTests.cpp b/llvm/unittests/Target/AMDGPU/AMDGPUUnitTests.cpp
index 51fda88e66ecac..33b4eca0a1dd30 100644
--- a/llvm/unittests/Target/AMDGPU/AMDGPUUnitTests.cpp
+++ b/llvm/unittests/Target/AMDGPU/AMDGPUUnitTests.cpp
@@ -276,6 +276,23 @@ TEST_F(AMDGPUTestBase, TestOccupancyAbsoluteLimits) {
                      256);
 }
 
+TEST_F(AMDGPUTestBase, TestGFX6LocalMemorySize) {
+  for (StringRef CPU : {"gfx600", "gfx601", "gfx602"}) {
+    SCOPED_TRACE(CPU.str());
+    auto TM = createAMDGPUTargetMachine(Triple("amdgcn-amd-amdhsa"), CPU, "");
+    ASSERT_NE(TM, nullptr);
+    GCNSubtarget ST(TM->getTargetTriple(), CPU.str(), "", *TM);
+
+    EXPECT_EQ(ST.getLocalMemorySize(), 65536u);
+    EXPECT_EQ(ST.getAddressableLocalMemorySize(), 32768u);
+
+    // Two workgroups using 32 KiB each fit in a CU. With 256 threads per
+    // workgroup, each SIMD can therefore hold two waves, limited only by LDS.
+    EXPECT_EQ(ST.getOccupancyWithWorkGroupSizes(32768, {256, 256}),
+              std::make_pair(2u, 2u));
+  }
+}
+
 static const char *printSubReg(const TargetRegisterInfo &TRI, unsigned SubReg) {
   return SubReg ? TRI.getSubRegIndexName(SubReg) : "<none>";
 }
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 0037bc63e7b0e2..550185ebe820de 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -2867,7 +2867,11 @@ TEST(TargetParserTest, testAMDGPUHalfAddressableLDSFeature) {
         AMDGPU::FEAT_HALF_ADDRESSABLE_PHYSICAL_LOCAL_MEMORY);
   };
 
-  // Only gfx10/11/12 address half of the physical LDS block.
+  // Gfx6 and gfx10/11/12 address half of the physical LDS block.
+  EXPECT_TRUE(Has(AMDGPU::GK_GFX600));
+  EXPECT_TRUE(Has(AMDGPU::GK_GFX601));
+  EXPECT_TRUE(Has(AMDGPU::GK_GFX602));
+  EXPECT_FALSE(Has(AMDGPU::GK_GFX700));
   EXPECT_FALSE(Has(AMDGPU::GK_GFX900));
   EXPECT_TRUE(Has(AMDGPU::GK_GFX1030));
   EXPECT_TRUE(Has(AMDGPU::GK_GFX1100));

``````````

</details>


https://github.com/llvm/llvm-project/pull/223185


More information about the llvm-commits mailing list