[llvm] [AMDGPU] Add `getLocalMemorySize` and `getAddressableLocalMemorySize` to TargetParser (PR #220718)

Chinmay Deshpande via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 11 11:53:04 PDT 2026


https://github.com/chinmaydd updated https://github.com/llvm/llvm-project/pull/220718

>From 2c714a177e2e8767ba4c32cf3c84c9d79192ea12 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Wed, 2 Sep 2026 16:02:17 -0400
Subject: [PATCH 1/2] [AMDGPU] Add getLocalMemorySize to TargetParser

Add getLocalMemorySize and getAddressableLocalMemorySize, both taking a
GPUKind or a Triple::SubArchType, so the LDS a work-group gets can be
queried from a GPU name alone without an MCSubtargetInfo. The first
returns the physical block available in the current mode, the second
caps it at what one work-group can address, mirroring the IsaInfo pair.

The number of SIMDs a work-group runs on is a per-kernel mode rather
than a property of the GPU, so it stays a parameter. GCNSubtarget
initializes its cached sizes from the new entry points. There is no
functional change.

Change-Id: Ib71428b66032a231ed491d6294122b359abfe7e2
Co-Authored-By: Claude Opus 5 (1M context) <noreply at anthropic.com>
---
 .../llvm/TargetParser/AMDGPUTargetParser.h    | 42 +++++++++++++-
 llvm/lib/Target/AMDGPU/GCNSubtarget.cpp       |  7 ++-
 llvm/lib/TargetParser/AMDGPUTargetParser.cpp  | 30 ++++++++++
 .../TargetParser/TargetParserTest.cpp         | 56 +++++++++++++++++++
 4 files changed, 129 insertions(+), 6 deletions(-)

diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index 3f74320675d12..efc3cc7151de9 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -192,13 +192,49 @@ LLVM_ABI unsigned getAddressableNumVGPRs(GPUKind AK, bool IsWave32);
 LLVM_ABI unsigned getAddressableNumVGPRs(Triple::SubArchType SubArch,
                                          bool IsWave32);
 
-/// \returns Maximum LDS in bytes a single work-group can address. This is a
-/// fixed hardware cap and does not depend on how many SIMDs a work-group runs
-/// on.
+/// LDS size queries.
+///
+/// \c getMaxHWAddressableLocalMemorySize returns the architectural limit that
+/// one work-group can address. It is independent of execution mode.
+///
+/// \c getLocalMemorySize returns the LDS available to all work-groups sharing a
+/// WGP or CU, which is the LDS capacity used to compute occupancy. In full-SIMD
+/// mode, a work-group runs on four SIMDs and the query returns the full
+/// physical block. In CU mode, it runs on two SIMDs and the query returns half
+/// the block.
+///
+/// \c getAddressableLocalMemorySize returns the amount one work-group can
+/// allocate:
+///
+///   min(getMaxHWAddressableLocalMemorySize(), getLocalMemorySize())
+///
+/// The physical LDS block belongs to a WGP on gfx10/11/12 and to a CU
+/// otherwise. On gfx10/11/12, the block is twice the address limit, so a
+/// work-group cannot address the entire block in WGP mode.
+///
+/// The mode columns below show local/addressable LDS, in KiB:
+///
+///   GPU      address limit   full-SIMD   CU mode
+///   gfx900              64        64/64   n/a (no CU mode)
+///   gfx1030             64       128/64   64/64
+///   gfx1250            320      320/320   n/a (always full-SIMD)
+
+/// \returns Maximum LDS in bytes a single work-group can address.
 LLVM_ABI unsigned getMaxHWAddressableLocalMemorySize(GPUKind AK);
 LLVM_ABI unsigned
 getMaxHWAddressableLocalMemorySize(Triple::SubArchType SubArch);
 
+/// \returns Total LDS in bytes available on one WGP or CU. \p FullSIMDMode
+/// selects whether a work-group runs on all four SIMDs.
+LLVM_ABI unsigned getLocalMemorySize(GPUKind AK, bool FullSIMDMode);
+LLVM_ABI unsigned getLocalMemorySize(Triple::SubArchType SubArch,
+                                     bool FullSIMDMode);
+
+/// \returns LDS in bytes a single work-group can allocate.
+LLVM_ABI unsigned getAddressableLocalMemorySize(GPUKind AK, bool FullSIMDMode);
+LLVM_ABI unsigned getAddressableLocalMemorySize(Triple::SubArchType SubArch,
+                                                bool FullSIMDMode);
+
 /// \returns Number of LDS banks per compute unit.
 LLVM_ABI unsigned getLDSBankCount(GPUKind AK);
 LLVM_ABI unsigned getLDSBankCount(Triple::SubArchType SubArch);
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
index 9443420c3d424..cbf2b35ebba7b 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
@@ -181,9 +181,10 @@ GCNSubtarget &GCNSubtarget::initializeSubtargetDependencies(const Triple &TT,
   if (FlatOffsetBitWidth == 0)
     FlatOffsetBitWidth = 13;
 
-  LocalMemorySize = AMDGPU::IsaInfo::getLocalMemorySize(*this);
-  AddressableLocalMemorySize =
-      AMDGPU::IsaInfo::getAddressableLocalMemorySize(*this);
+  LocalMemorySize = AMDGPU::getLocalMemorySize(getTargetID().getGPUKind(),
+                                               AMDGPU::isFullSIMDMode(*this));
+  AddressableLocalMemorySize = AMDGPU::getAddressableLocalMemorySize(
+      getTargetID().getGPUKind(), AMDGPU::isFullSIMDMode(*this));
   // LDS Allocation Granularity calculated in bytes from dwords
   LDSAllocationGranularity =
       AMDGPU::getLdsDwGranularity(*this) * sizeof(uint32_t);
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index bf6acf0774215..1f0f44c22de2e 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -16,6 +16,7 @@
 #include "llvm/ADT/Twine.h"
 #include "llvm/Support/raw_ostream.h"
 #include "llvm/TargetParser/Triple.h"
+#include <algorithm>
 #include <array>
 #include <cassert>
 
@@ -476,6 +477,35 @@ AMDGPU::getMaxHWAddressableLocalMemorySize(Triple::SubArchType SubArch) {
   return getMaxHWAddressableLocalMemorySize(getGPUKindFromSubArch(SubArch));
 }
 
+unsigned AMDGPU::getLocalMemorySize(GPUKind AK, bool FullSIMDMode) {
+  // gfx10/11/12 address half of the physical block, e.g. 64 KiB of 128 KiB.
+  unsigned Size = getMaxHWAddressableLocalMemorySize(AK);
+  if (getFeatureBitset(AK).test(FEAT_HALF_ADDRESSABLE_PHYSICAL_LOCAL_MEMORY))
+    Size *= 2;
+
+  // In CU mode the work-group runs on two SIMDs and reaches only their half.
+  if (!FullSIMDMode)
+    Size /= 2;
+
+  return Size;
+}
+
+unsigned AMDGPU::getLocalMemorySize(Triple::SubArchType SubArch,
+                                    bool FullSIMDMode) {
+  return getLocalMemorySize(getGPUKindFromSubArch(SubArch), FullSIMDMode);
+}
+
+unsigned AMDGPU::getAddressableLocalMemorySize(GPUKind AK, bool FullSIMDMode) {
+  return std::min(getMaxHWAddressableLocalMemorySize(AK),
+                  getLocalMemorySize(AK, FullSIMDMode));
+}
+
+unsigned AMDGPU::getAddressableLocalMemorySize(Triple::SubArchType SubArch,
+                                               bool FullSIMDMode) {
+  return getAddressableLocalMemorySize(getGPUKindFromSubArch(SubArch),
+                                       FullSIMDMode);
+}
+
 unsigned AMDGPU::getLDSBankCount(GPUKind AK) {
   const GPUInfo *Info = getAMDGPUInfo(AK);
   return Info ? Info->LDSBankCount : 32;
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index cbf3d86c48b13..e24484737a54a 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -3220,6 +3220,62 @@ TEST(TargetParserTest, testAMDGPUgetMaxHWAddressableLocalMemorySize) {
             327680u);
 }
 
+TEST(TargetParserTest, testAMDGPUgetLocalMemorySize) {
+  // Without a half-addressable physical block the total matches the
+  // addressable cap, and running on two SIMDs halves it.
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(AMDGPU::GK_GFX600, true), 32768u);
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(AMDGPU::GK_GFX600, false), 16384u);
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(AMDGPU::GK_GFX900, true), 65536u);
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(AMDGPU::GK_GFX950, true), 163840u);
+
+  // gfx10/11/12 address 64 KiB of a 128 KiB block.
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(AMDGPU::GK_GFX1030, true), 131072u);
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(AMDGPU::GK_GFX1030, false), 65536u);
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(AMDGPU::GK_GFX1100, true), 131072u);
+
+  // gfx12.5 and gfx13 dropped the half-addressable block.
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(AMDGPU::GK_GFX1250, true), 327680u);
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(AMDGPU::GK_GFX1310, true), 196608u);
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(AMDGPU::GK_GFX1310, false), 98304u);
+
+  // An unknown GPU falls back to the smallest block.
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(AMDGPU::GK_NONE, true), 32768u);
+
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(Triple::AMDGPUSubArch900, true), 65536u);
+  EXPECT_EQ(AMDGPU::getLocalMemorySize(Triple::AMDGPUSubArch1030, true),
+            131072u);
+}
+
+TEST(TargetParserTest, testAMDGPUgetAddressableLocalMemorySize) {
+  // A work-group never allocates past the hardware cap, so the doubled
+  // gfx10/11/12 block is capped back to the addressable size.
+  EXPECT_EQ(AMDGPU::getAddressableLocalMemorySize(AMDGPU::GK_GFX1030, true),
+            65536u);
+  EXPECT_EQ(AMDGPU::getAddressableLocalMemorySize(AMDGPU::GK_GFX1030, false),
+            65536u);
+  EXPECT_EQ(AMDGPU::getAddressableLocalMemorySize(AMDGPU::GK_GFX1100, true),
+            65536u);
+
+  // Without a doubled block the cap is only reached in full-SIMD mode.
+  EXPECT_EQ(AMDGPU::getAddressableLocalMemorySize(AMDGPU::GK_GFX600, true),
+            32768u);
+  EXPECT_EQ(AMDGPU::getAddressableLocalMemorySize(AMDGPU::GK_GFX600, false),
+            16384u);
+  EXPECT_EQ(AMDGPU::getAddressableLocalMemorySize(AMDGPU::GK_GFX950, true),
+            163840u);
+  EXPECT_EQ(AMDGPU::getAddressableLocalMemorySize(AMDGPU::GK_GFX1250, true),
+            327680u);
+  EXPECT_EQ(AMDGPU::getAddressableLocalMemorySize(AMDGPU::GK_GFX1310, false),
+            98304u);
+
+  EXPECT_EQ(AMDGPU::getAddressableLocalMemorySize(AMDGPU::GK_NONE, true),
+            32768u);
+
+  EXPECT_EQ(
+      AMDGPU::getAddressableLocalMemorySize(Triple::AMDGPUSubArch1030, true),
+      65536u);
+}
+
 TEST(TargetParserTest, testAMDGPUgetNumWorkGroupSIMDs) {
   EXPECT_EQ(AMDGPU::getNumWorkGroupSIMDs(true), 4u);
   EXPECT_EQ(AMDGPU::getNumWorkGroupSIMDs(false), 2u);

>From 49ea2edc1c4a5cb0732eb0659aaa943f1caf6aee Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Thu, 10 Sep 2026 17:30:54 -0400
Subject: [PATCH 2/2] [AMDGPU] Use SIMD-mode terminology for LDS queries

Change-Id: I6d6900868d58374d295126c52f67ffcdd8676902
---
 .../llvm/TargetParser/AMDGPUTargetParser.h    | 19 ++++++++++---------
 llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h      |  3 +++
 llvm/lib/Target/AMDGPU/GCNSubtarget.cpp       |  9 ++++-----
 llvm/lib/TargetParser/AMDGPUTargetParser.cpp  |  2 +-
 4 files changed, 18 insertions(+), 15 deletions(-)

diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index efc3cc7151de9..658dfee0544a1 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -198,10 +198,10 @@ LLVM_ABI unsigned getAddressableNumVGPRs(Triple::SubArchType SubArch,
 /// one work-group can address. It is independent of execution mode.
 ///
 /// \c getLocalMemorySize returns the LDS available to all work-groups sharing a
-/// WGP or CU, which is the LDS capacity used to compute occupancy. In full-SIMD
-/// mode, a work-group runs on four SIMDs and the query returns the full
-/// physical block. In CU mode, it runs on two SIMDs and the query returns half
-/// the block.
+/// physical block, which is the LDS capacity used to compute occupancy. In
+/// full-SIMD mode, a work-group runs on four SIMDs and the query returns the
+/// full physical block. In half-SIMD mode, it runs on two SIMDs and the query
+/// returns half the block.
 ///
 /// \c getAddressableLocalMemorySize returns the amount one work-group can
 /// allocate:
@@ -210,12 +210,12 @@ LLVM_ABI unsigned getAddressableNumVGPRs(Triple::SubArchType SubArch,
 ///
 /// The physical LDS block belongs to a WGP on gfx10/11/12 and to a CU
 /// otherwise. On gfx10/11/12, the block is twice the address limit, so a
-/// work-group cannot address the entire block in WGP mode.
+/// work-group cannot address the entire block in full-SIMD mode.
 ///
 /// The mode columns below show local/addressable LDS, in KiB:
 ///
-///   GPU      address limit   full-SIMD   CU mode
-///   gfx900              64        64/64   n/a (no CU mode)
+///   GPU      address limit   full-SIMD   half-SIMD
+///   gfx900              64        64/64   n/a (always full-SIMD)
 ///   gfx1030             64       128/64   64/64
 ///   gfx1250            320      320/320   n/a (always full-SIMD)
 
@@ -224,8 +224,9 @@ LLVM_ABI unsigned getMaxHWAddressableLocalMemorySize(GPUKind AK);
 LLVM_ABI unsigned
 getMaxHWAddressableLocalMemorySize(Triple::SubArchType SubArch);
 
-/// \returns Total LDS in bytes available on one WGP or CU. \p FullSIMDMode
-/// selects whether a work-group runs on all four SIMDs.
+/// \returns Total LDS in bytes available to work-groups sharing a physical
+/// block. \p FullSIMDMode selects full-SIMD mode (four SIMDs) when true and
+/// half-SIMD mode (two SIMDs) otherwise.
 LLVM_ABI unsigned getLocalMemorySize(GPUKind AK, bool FullSIMDMode);
 LLVM_ABI unsigned getLocalMemorySize(Triple::SubArchType SubArch,
                                      bool FullSIMDMode);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
index e1f331229c463..5cb4e1a23d262 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
@@ -242,6 +242,9 @@ class AMDGPUSubtarget {
   /// SIMDs in full-SIMD mode, half of them otherwise.
   unsigned getNumWorkGroupSIMDs() const { return NumWorkGroupSIMDs; }
 
+  /// \returns Whether a work-group runs on all of the block's SIMDs.
+  bool isFullSIMDMode() const { return NumWorkGroupSIMDs == 4; }
+
   Align getAlignmentForImplicitArgPtr() const {
     return isAmdHsaOS() ? Align(8) : Align(4);
   }
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
index cbf2b35ebba7b..d0706e26da8a7 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
@@ -181,10 +181,12 @@ GCNSubtarget &GCNSubtarget::initializeSubtargetDependencies(const Triple &TT,
   if (FlatOffsetBitWidth == 0)
     FlatOffsetBitWidth = 13;
 
+  NumWorkGroupSIMDs =
+      AMDGPU::getNumWorkGroupSIMDs(AMDGPU::isFullSIMDMode(*this));
   LocalMemorySize = AMDGPU::getLocalMemorySize(getTargetID().getGPUKind(),
-                                               AMDGPU::isFullSIMDMode(*this));
+                                               isFullSIMDMode());
   AddressableLocalMemorySize = AMDGPU::getAddressableLocalMemorySize(
-      getTargetID().getGPUKind(), AMDGPU::isFullSIMDMode(*this));
+      getTargetID().getGPUKind(), isFullSIMDMode());
   // LDS Allocation Granularity calculated in bytes from dwords
   LDSAllocationGranularity =
       AMDGPU::getLdsDwGranularity(*this) * sizeof(uint32_t);
@@ -250,9 +252,6 @@ GCNSubtarget::GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS,
   LLVM_DEBUG(dbgs() << "sramecc setting for subtarget: "
                     << TargetID.getSramEccSetting() << '\n');
 
-  NumWorkGroupSIMDs =
-      AMDGPU::getNumWorkGroupSIMDs(AMDGPU::isFullSIMDMode(*this));
-
   TSInfo = std::make_unique<AMDGPUSelectionDAGInfo>();
 
   CallLoweringInfo = std::make_unique<AMDGPUCallLowering>(*getTargetLowering());
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index 1f0f44c22de2e..4f5acad83ab2f 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -483,7 +483,7 @@ unsigned AMDGPU::getLocalMemorySize(GPUKind AK, bool FullSIMDMode) {
   if (getFeatureBitset(AK).test(FEAT_HALF_ADDRESSABLE_PHYSICAL_LOCAL_MEMORY))
     Size *= 2;
 
-  // In CU mode the work-group runs on two SIMDs and reaches only their half.
+  // In half-SIMD mode the work-group reaches only half of the block.
   if (!FullSIMDMode)
     Size /= 2;
 



More information about the llvm-commits mailing list