[llvm-branch-commits] [llvm] [AMDGPU] Move `getAddressableNumVGPRs` to TargetParser (PR #226821)

Chinmay Deshpande via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Sun Sep 27 11:30:02 PDT 2026


https://github.com/chinmaydd created https://github.com/llvm/llvm-project/pull/226821

Extend getAddressableNumVGPRs with a dynamic block size and share the eight-block limit through TargetParser. Remove the BaseInfo counterpart and update its subtarget, occupancy, scheduler, and diagnostic callers.

>From 45dde814328f9ecf3cd957719277b7dafe61f698 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sun, 27 Sep 2026 14:15:39 -0400
Subject: [PATCH] [AMDGPU] Move dynamic VGPR addressability to TargetParser

Extend getAddressableNumVGPRs with a dynamic block size and share the
eight-block limit through TargetParser. Remove the BaseInfo counterpart
and update its subtarget, occupancy, scheduler, and diagnostic callers.

Cover dynamic addressability, static limits, and unified register files.

Change-Id: Ie4dae7ba34cce8e8ea50e24f7075242a8a045505
---
 .../llvm/TargetParser/AMDGPUTargetParser.h    | 15 ++++++---
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp   | 13 ++++----
 llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp   |  3 +-
 llvm/lib/Target/AMDGPU/GCNSubtarget.h         |  8 +----
 .../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp    | 17 ++--------
 llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h |  7 -----
 llvm/lib/TargetParser/AMDGPUTargetParser.cpp  | 11 +++++--
 .../llvm-calc-occupancy.cpp                   |  3 +-
 .../TargetParser/TargetParserTest.cpp         | 31 +++++++++++++++++++
 9 files changed, 61 insertions(+), 47 deletions(-)

diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index d8ad8dcd415c3..e7f758deb2cae 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -196,12 +196,19 @@ LLVM_ABI unsigned getVGPREncodingGranule(Triple::SubArchType SubArch,
 LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32);
 LLVM_ABI unsigned getTotalNumVGPRs(Triple::SubArchType SubArch, bool IsWave32);
 
+/// Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
+constexpr unsigned MaxDynamicVGPRBlocks = 8;
+
 /// \returns Number of VGPRs a single wave can address. On a target with a
-/// unified register file this covers the AGPRs as well. This does not account
-/// for dynamic VGPR mode, which caps allocation at a fixed number of blocks.
-LLVM_ABI unsigned getAddressableNumVGPRs(GPUKind AK, bool IsWave32);
+/// unified register file this covers the AGPRs as well. A nonzero
+/// \p DynamicVGPRBlockSize selects dynamic VGPR mode, which caps allocation at
+/// \c MaxDynamicVGPRBlocks blocks. On gfx90a-family targets, the unified
+/// register file size is returned regardless of \p DynamicVGPRBlockSize.
+LLVM_ABI unsigned getAddressableNumVGPRs(GPUKind AK, bool IsWave32,
+                                         unsigned DynamicVGPRBlockSize = 0);
 LLVM_ABI unsigned getAddressableNumVGPRs(Triple::SubArchType SubArch,
-                                         bool IsWave32);
+                                         bool IsWave32,
+                                         unsigned DynamicVGPRBlockSize = 0);
 
 /// LDS size queries.
 ///
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index cbe54f4d89832..d038f1a442f64 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1161,13 +1161,12 @@ void AMDGPUAsmPrinter::emitDVgprSymbol(MachineFunction &MF) {
       unsigned NumBlocks =
           divideCeil(std::max(unsigned(NumVGPRs.getConstant()), 1U), BlockSize);
 
-      if (NumBlocks > AMDGPU::IsaInfo::MaxDynamicVGPRBlocks) {
-        OutContext.reportError(
-            {}, "DVGPR block count " + Twine(NumBlocks) +
-                    " exceeds maximum of " +
-                    Twine(AMDGPU::IsaInfo::MaxDynamicVGPRBlocks) +
-                    " for __dvgpr$ symbol for '" +
-                    Twine(CurrentFnSym->getName()) + "'");
+      if (NumBlocks > AMDGPU::MaxDynamicVGPRBlocks) {
+        OutContext.reportError({}, "DVGPR block count " + Twine(NumBlocks) +
+                                       " exceeds maximum of " +
+                                       Twine(AMDGPU::MaxDynamicVGPRBlocks) +
+                                       " for __dvgpr$ symbol for '" +
+                                       Twine(CurrentFnSym->getName()) + "'");
         return;
       }
       unsigned EncodedNumBlocks = (NumBlocks - 1) << 3;
diff --git a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
index 460c872dc46d6..fd7d643404e39 100644
--- a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
@@ -170,8 +170,7 @@ void GCNSchedStrategy::initialize(ScheduleDAGMI *DAG) {
                          "VGPRCriticalLimit calculation method.\n");
     unsigned DynamicVGPRBlockSize = MFI.getDynamicVGPRBlockSize();
     unsigned Granule = ST.getVGPRAllocGranule(DynamicVGPRBlockSize);
-    unsigned Addressable =
-        AMDGPU::IsaInfo::getAddressableNumVGPRs(ST, DynamicVGPRBlockSize);
+    unsigned Addressable = ST.getAddressableNumVGPRs(DynamicVGPRBlockSize);
     unsigned VGPRBudget = alignDown(Addressable / TargetOccupancy, Granule);
     VGPRBudget = std::max(VGPRBudget, Granule);
     VGPRCriticalLimit = std::min(VGPRBudget, VGPRExcessLimit);
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 437cc9e942d5a..d5660a2c9fb04 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -871,14 +871,8 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
 
   /// \returns Addressable number of VGPRs supported by the subtarget.
   unsigned getAddressableNumVGPRs(unsigned DynamicVGPRBlockSize) const {
-    // Dynamic VGPR mode is a per-kernel mode, so it is not covered by the
-    // TargetParser query.
-    if (DynamicVGPRBlockSize != 0) {
-      return AMDGPU::IsaInfo::getAddressableNumVGPRs(*this,
-                                                     DynamicVGPRBlockSize);
-    }
     return AMDGPU::getAddressableNumVGPRs(getTargetID().getGPUKind(),
-                                          isWave32());
+                                          isWave32(), DynamicVGPRBlockSize);
   }
 
   /// \returns the minimum number of VGPRs that will prevent achieving more than
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 82a9603915291..63ece72e08879 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1307,19 +1307,6 @@ unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI) {
   return 256;
 }
 
-unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
-                                unsigned DynamicVGPRBlockSize) {
-  const auto &Features = STI.getFeatureBits();
-  if (Features.test(FeatureGFX90AInsts))
-    return 512;
-
-  if (DynamicVGPRBlockSize != 0) {
-    // On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
-    return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
-  }
-  return getAddressableNumArchVGPRs(STI);
-}
-
 unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI,
                                       unsigned NumVGPRs,
                                       unsigned DynamicVGPRBlockSize) {
@@ -1382,7 +1369,7 @@ unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
   bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
   unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(Kind, IsWave32);
   unsigned AddrsableNumVGPRs =
-      getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
+      AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, DynamicVGPRBlockSize);
   unsigned Granule =
       AMDGPU::getVGPRAllocGranule(Kind, IsWave32, DynamicVGPRBlockSize);
   unsigned MaxNumVGPRs = alignDown(TotNumVGPRs / WavesPerEU, Granule);
@@ -1416,7 +1403,7 @@ unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
                                      AMDGPU::getVGPRAllocGranule(
                                          Kind, IsWave32, DynamicVGPRBlockSize));
   unsigned AddressableNumVGPRs =
-      getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
+      AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, DynamicVGPRBlockSize);
   return std::min(MaxNumVGPRs, AddressableNumVGPRs);
 }
 
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index a57cfee284738..211771a8dab11 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -237,17 +237,10 @@ unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs);
 /// returns the allocation granule for ArchVGPRs.
 unsigned getArchVGPRAllocGranule();
 
-/// Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
-static constexpr unsigned MaxDynamicVGPRBlocks = 8;
-
 /// \returns Addressable number of architectural VGPRs for a given subtarget \p
 /// STI.
 unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI);
 
-/// \returns Addressable number of VGPRs for given subtarget \p STI.
-unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
-                                unsigned DynamicVGPRBlockSize);
-
 /// \returns Minimum number of VGPRs that meets given number of waves per
 /// execution unit requirement for given subtarget \p STI.
 unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index bcdf5c356ef98..d82a82bd9699e 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -471,19 +471,24 @@ unsigned AMDGPU::getTotalNumVGPRs(Triple::SubArchType SubArch, bool IsWave32) {
   return getTotalNumVGPRs(getGPUKindFromSubArch(SubArch), IsWave32);
 }
 
-unsigned AMDGPU::getAddressableNumVGPRs(GPUKind AK, bool IsWave32) {
+unsigned AMDGPU::getAddressableNumVGPRs(GPUKind AK, bool IsWave32,
+                                        unsigned DynamicVGPRBlockSize) {
   const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
   // The unified register file makes the AGPRs addressable as VGPRs.
   if (Features.test(FEAT_GFX90A_INSTS))
     return 512;
+  if (DynamicVGPRBlockSize != 0)
+    return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
   if (Features.test(FEAT_1024_ADDRESSABLE_VGPRS))
     return IsWave32 ? 1024 : 512;
   return 256;
 }
 
 unsigned AMDGPU::getAddressableNumVGPRs(Triple::SubArchType SubArch,
-                                        bool IsWave32) {
-  return getAddressableNumVGPRs(getGPUKindFromSubArch(SubArch), IsWave32);
+                                        bool IsWave32,
+                                        unsigned DynamicVGPRBlockSize) {
+  return getAddressableNumVGPRs(getGPUKindFromSubArch(SubArch), IsWave32,
+                                DynamicVGPRBlockSize);
 }
 
 unsigned AMDGPU::getMaxHWAddressableLocalMemorySize(GPUKind AK) {
diff --git a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
index a21165d041bc6..2ef08e4bb70d1 100644
--- a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
+++ b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
@@ -218,8 +218,7 @@ int main(int argc, char **argv) {
   unsigned NumWorkGroupSIMDs = ST.getNumWorkGroupSIMDs();
   unsigned LocalMemSize = AMDGPU::IsaInfo::getLocalMemorySize(STI);
   unsigned AddrLocalMem = AMDGPU::IsaInfo::getAddressableLocalMemorySize(STI);
-  unsigned AddrVGPRs =
-      AMDGPU::IsaInfo::getAddressableNumVGPRs(STI, DynVGPRBlockSize);
+  unsigned AddrVGPRs = ST.getAddressableNumVGPRs(DynVGPRBlockSize);
   unsigned AddrSGPRs = ST.getAddressableNumSGPRs();
   unsigned MaxWGSize = AMDGPU::getMaxFlatWorkGroupSize();
 
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 7c5f156775bd2..1844f929fd849 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -3327,6 +3327,37 @@ TEST(TargetParserTest, testAMDGPUDynamicVGPRAllocGranule) {
             8u);
 }
 
+TEST(TargetParserTest, testAMDGPUDynamicVGPRAddressableNum) {
+  for (auto Kind :
+       {AMDGPU::GK_GFX1200, AMDGPU::GK_GFX1250, AMDGPU::GK_GFX1310}) {
+    SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+    for (bool IsWave32 : {false, true}) {
+      EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, 16), 128u);
+      EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, 32), 256u);
+    }
+  }
+
+  // Zero selects the static limits, which depend on the wavefront size.
+  EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(AMDGPU::GK_GFX1250, false, 0), 512u);
+  EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(AMDGPU::GK_GFX1250, true, 0), 1024u);
+
+  // gfx90a-family targets always use their unified register file size.
+  for (auto Kind : {AMDGPU::GK_GFX90A, AMDGPU::GK_GFX942, AMDGPU::GK_GFX950}) {
+    SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+    for (bool IsWave32 : {false, true}) {
+      EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, 16), 512u);
+    }
+  }
+
+  EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Triple::AMDGPUSubArch1200, true, 16),
+            128u);
+  EXPECT_EQ(
+      AMDGPU::getAddressableNumVGPRs(Triple::AMDGPUSubArch1250, false, 32),
+      256u);
+  EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Triple::AMDGPUSubArch90A, false, 16),
+            512u);
+}
+
 TEST(TargetParserTest, testAMDGPUgetMaxHWAddressableLocalMemorySize) {
   // The addressable cap is a fixed hardware property, independent of how many
   // SIMDs a work-group runs on.



More information about the llvm-branch-commits mailing list