[llvm-branch-commits] [llvm] [AMDGPU] Move dynamic VGPR allocation granules to TargetParser (PR #226818)

Chinmay Deshpande via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Sun Sep 27 11:25:13 PDT 2026


https://github.com/chinmaydd created https://github.com/llvm/llvm-project/pull/226818

Extend getVGPRAllocGranule with a dynamic block size and preserve the fixed gfx90a-family granule. Remove the BaseInfo counterpart and migrate its occupancy, scheduler, frame lowering, and register-block callers.

>From de480cc59fc39991e1fb2eff30b70bffb7aa8227 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sun, 27 Sep 2026 14:13:31 -0400
Subject: [PATCH] [AMDGPU] Move dynamic VGPR allocation granules to
 TargetParser

Extend getVGPRAllocGranule with a dynamic block size and preserve the
fixed gfx90a-family granule. Remove the BaseInfo counterpart and migrate
its occupancy, scheduler, frame lowering, and register-block callers.

Cover dynamic allocation, static mode, and gfx90a-family exceptions.

Change-Id: Id441ddcea2d7a33121ddb4139af4c5f52f273ff3
---
 .../llvm/TargetParser/AMDGPUTargetParser.h    | 11 ++--
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp   |  2 +-
 llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp   |  3 +-
 llvm/lib/Target/AMDGPU/GCNSubtarget.h         |  3 +-
 llvm/lib/Target/AMDGPU/SIFrameLowering.cpp    |  3 +-
 .../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp    | 55 ++++++-------------
 llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h |  8 ---
 llvm/lib/TargetParser/AMDGPUTargetParser.cpp  | 12 ++--
 .../TargetParser/TargetParserTest.cpp         | 30 ++++++++++
 9 files changed, 68 insertions(+), 59 deletions(-)

diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index b611e5cbcc4fe..d8ad8dcd415c3 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -174,11 +174,14 @@ LLVM_ABI unsigned getSGPRAllocGranule(Triple::SubArchType SubArch);
 
 /// \returns VGPR allocation granularity for \p AK, in registers. \p IsWave32
 /// selects the wavefront size, which is a per-kernel mode rather than a
-/// property of the GPU. This does not account for dynamic VGPR mode, where the
-/// block size chosen by the caller is the granule.
-LLVM_ABI unsigned getVGPRAllocGranule(GPUKind AK, bool IsWave32);
+/// property of the GPU. A nonzero \p DynamicVGPRBlockSize selects dynamic
+/// VGPR mode, where the block size is the granule. On gfx90a-family targets,
+/// the fixed granule is used regardless of \p DynamicVGPRBlockSize.
+LLVM_ABI unsigned getVGPRAllocGranule(GPUKind AK, bool IsWave32,
+                                      unsigned DynamicVGPRBlockSize = 0);
 LLVM_ABI unsigned getVGPRAllocGranule(Triple::SubArchType SubArch,
-                                      bool IsWave32);
+                                      bool IsWave32,
+                                      unsigned DynamicVGPRBlockSize = 0);
 
 /// \returns VGPR encoding granularity for \p AK, in registers. \p IsWave32
 /// selects the wavefront size encoded in the kernel descriptor. This can
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 89105a78c4e80..cbe54f4d89832 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -438,7 +438,7 @@ const AMDGPUMCExpr *createOccupancy(unsigned InitOcc, const MCExpr *NumSGPRs,
                                     unsigned DynamicVGPRBlockSize,
                                     const GCNSubtarget &STM, MCContext &Ctx) {
   unsigned MaxWaves = STM.getMaxWavesPerEU();
-  unsigned Granule = IsaInfo::getVGPRAllocGranule(STM, DynamicVGPRBlockSize);
+  unsigned Granule = STM.getVGPRAllocGranule(DynamicVGPRBlockSize);
   unsigned TargetTotalNumVGPRs = STM.getTotalNumVGPRs();
 
   // Bake the per-function SGPR budget into the operands so the late-evaluated
diff --git a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
index 602489ab3a5c1..460c872dc46d6 100644
--- a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
@@ -169,8 +169,7 @@ void GCNSchedStrategy::initialize(ScheduleDAGMI *DAG) {
     LLVM_DEBUG(dbgs() << "Region is known to spill, use alternative "
                          "VGPRCriticalLimit calculation method.\n");
     unsigned DynamicVGPRBlockSize = MFI.getDynamicVGPRBlockSize();
-    unsigned Granule =
-        AMDGPU::IsaInfo::getVGPRAllocGranule(ST, DynamicVGPRBlockSize);
+    unsigned Granule = ST.getVGPRAllocGranule(DynamicVGPRBlockSize);
     unsigned Addressable =
         AMDGPU::IsaInfo::getAddressableNumVGPRs(ST, DynamicVGPRBlockSize);
     unsigned VGPRBudget = alignDown(Addressable / TargetOccupancy, Granule);
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index c344367b460f4..437cc9e942d5a 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -848,7 +848,8 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
 
   /// \returns VGPR allocation granularity supported by the subtarget.
   unsigned getVGPRAllocGranule(unsigned DynamicVGPRBlockSize) const {
-    return AMDGPU::IsaInfo::getVGPRAllocGranule(*this, DynamicVGPRBlockSize);
+    return AMDGPU::getVGPRAllocGranule(getTargetID().getGPUKind(), isWave32(),
+                                       DynamicVGPRBlockSize);
   }
 
   /// \returns VGPR encoding granularity supported by the subtarget.
diff --git a/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp b/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
index 364d8f07b6df2..81a371352afcc 100644
--- a/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
@@ -889,8 +889,7 @@ void SIFrameLowering::emitEntryFunctionPrologue(MachineFunction &MF,
     assert(FPReg != AMDGPU::FP_REG);
     unsigned VGPRSize = llvm::alignTo(
         (ST.getAddressableNumVGPRs(MFI->getDynamicVGPRBlockSize()) -
-         AMDGPU::IsaInfo::getVGPRAllocGranule(ST,
-                                              MFI->getDynamicVGPRBlockSize())) *
+         ST.getVGPRAllocGranule(MFI->getDynamicVGPRBlockSize())) *
             4,
         FrameInfo.getMaxAlign());
     MFI->setScratchReservedForDynamicVGPRs(VGPRSize);
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index baaef6293b9b0..82a9603915291 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1298,28 +1298,6 @@ unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs) {
          1;
 }
 
-unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI,
-                             unsigned DynamicVGPRBlockSize,
-                             std::optional<bool> EnableWavefrontSize32) {
-  if (STI.getFeatureBits().test(FeatureGFX90AInsts))
-    return 8;
-
-  if (DynamicVGPRBlockSize != 0)
-    return DynamicVGPRBlockSize;
-
-  bool IsWave32 = EnableWavefrontSize32
-                      ? *EnableWavefrontSize32
-                      : STI.getFeatureBits().test(FeatureWavefrontSize32);
-
-  if (STI.getFeatureBits().test(Feature1536VGPRs))
-    return IsWave32 ? 24 : 12;
-
-  if (hasGFX10_3Insts(STI))
-    return IsWave32 ? 16 : 8;
-
-  return IsWave32 ? 8 : 4;
-}
-
 unsigned getArchVGPRAllocGranule() { return 4; }
 
 unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI) {
@@ -1337,8 +1315,7 @@ unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
 
   if (DynamicVGPRBlockSize != 0) {
     // On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
-    return MaxDynamicVGPRBlocks *
-           getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
+    return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
   }
   return getAddressableNumArchVGPRs(STI);
 }
@@ -1349,7 +1326,8 @@ unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI,
   GPUKind Kind = parseArchAMDGCN(STI.getCPU());
   bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
   return getNumWavesPerEUWithNumVGPRs(
-      NumVGPRs, getVGPRAllocGranule(STI, DynamicVGPRBlockSize),
+      NumVGPRs,
+      AMDGPU::getVGPRAllocGranule(Kind, IsWave32, DynamicVGPRBlockSize),
       getMaxWavesPerEU(Kind), AMDGPU::getTotalNumVGPRs(Kind, IsWave32));
 }
 
@@ -1401,11 +1379,12 @@ unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
   if (WavesPerEU >= MaxWavesPerEU)
     return 0;
 
-  unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(
-      Kind, STI.getFeatureBits().test(FeatureWavefrontSize32));
+  bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
+  unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(Kind, IsWave32);
   unsigned AddrsableNumVGPRs =
       getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
-  unsigned Granule = getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
+  unsigned Granule =
+      AMDGPU::getVGPRAllocGranule(Kind, IsWave32, DynamicVGPRBlockSize);
   unsigned MaxNumVGPRs = alignDown(TotNumVGPRs / WavesPerEU, Granule);
 
   if (MaxNumVGPRs == alignDown(TotNumVGPRs / MaxWavesPerEU, Granule))
@@ -1425,17 +1404,17 @@ unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
                         unsigned DynamicVGPRBlockSize) {
   assert(WavesPerEU != 0);
 
-  unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(
-      parseArchAMDGCN(STI.getCPU()),
-      STI.getFeatureBits().test(FeatureWavefrontSize32));
+  GPUKind Kind = parseArchAMDGCN(STI.getCPU());
+  bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
+  unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(Kind, IsWave32);
 
   // In dynamic VGPR mode, WavesPerEU does not imply a VGPR limit.
   bool DynamicVGPREnabled = (DynamicVGPRBlockSize != 0);
   unsigned MaxNumVGPRs =
-      DynamicVGPREnabled
-          ? TotNumVGPRs
-          : alignDown(TotNumVGPRs / WavesPerEU,
-                      getVGPRAllocGranule(STI, DynamicVGPRBlockSize));
+      DynamicVGPREnabled ? TotNumVGPRs
+                         : alignDown(TotNumVGPRs / WavesPerEU,
+                                     AMDGPU::getVGPRAllocGranule(
+                                         Kind, IsWave32, DynamicVGPRBlockSize));
   unsigned AddressableNumVGPRs =
       getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
   return std::min(MaxNumVGPRs, AddressableNumVGPRs);
@@ -1455,9 +1434,11 @@ unsigned getAllocatedNumVGPRBlocks(const MCSubtargetInfo &STI,
                                    unsigned NumVGPRs,
                                    unsigned DynamicVGPRBlockSize,
                                    std::optional<bool> EnableWavefrontSize32) {
+  bool IsWave32 = EnableWavefrontSize32.value_or(
+      STI.getFeatureBits().test(FeatureWavefrontSize32));
   return getGranulatedNumRegisterBlocks(
-      NumVGPRs,
-      getVGPRAllocGranule(STI, DynamicVGPRBlockSize, EnableWavefrontSize32));
+      NumVGPRs, AMDGPU::getVGPRAllocGranule(parseArchAMDGCN(STI.getCPU()),
+                                            IsWave32, DynamicVGPRBlockSize));
 }
 } // end namespace IsaInfo
 
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index 564d77b0ad938..a57cfee284738 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -233,14 +233,6 @@ unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed,
 /// register counts.
 unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs);
 
-/// \returns VGPR allocation granularity for given subtarget \p STI.
-///
-/// For subtargets which support it, \p EnableWavefrontSize32 should match
-/// the ENABLE_WAVEFRONT_SIZE32 kernel descriptor field.
-unsigned
-getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize,
-                    std::optional<bool> EnableWavefrontSize32 = std::nullopt);
-
 /// For subtargets with a unified VGPR file and mixed ArchVGPR/AGPR usage,
 /// returns the allocation granule for ArchVGPRs.
 unsigned getArchVGPRAllocGranule();
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index 359b65363f2d1..bcdf5c356ef98 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -422,10 +422,13 @@ unsigned AMDGPU::getSGPRAllocGranule(Triple::SubArchType SubArch) {
   return 8;
 }
 
-unsigned AMDGPU::getVGPRAllocGranule(GPUKind AK, bool IsWave32) {
+unsigned AMDGPU::getVGPRAllocGranule(GPUKind AK, bool IsWave32,
+                                     unsigned DynamicVGPRBlockSize) {
   const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
   if (Features.test(FEAT_GFX90A_INSTS))
     return 8;
+  if (DynamicVGPRBlockSize != 0)
+    return DynamicVGPRBlockSize;
   if (Features.test(FEAT_1536_PHYSICAL_VGPRS))
     return IsWave32 ? 24 : 12;
   if (Features.test(FEAT_GFX10_3_INSTS))
@@ -433,9 +436,10 @@ unsigned AMDGPU::getVGPRAllocGranule(GPUKind AK, bool IsWave32) {
   return IsWave32 ? 8 : 4;
 }
 
-unsigned AMDGPU::getVGPRAllocGranule(Triple::SubArchType SubArch,
-                                     bool IsWave32) {
-  return getVGPRAllocGranule(getGPUKindFromSubArch(SubArch), IsWave32);
+unsigned AMDGPU::getVGPRAllocGranule(Triple::SubArchType SubArch, bool IsWave32,
+                                     unsigned DynamicVGPRBlockSize) {
+  return getVGPRAllocGranule(getGPUKindFromSubArch(SubArch), IsWave32,
+                             DynamicVGPRBlockSize);
 }
 
 unsigned AMDGPU::getVGPREncodingGranule(GPUKind AK, bool IsWave32) {
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 2c963026a6a45..7c5f156775bd2 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -3297,6 +3297,36 @@ TEST(TargetParserTest, testAMDGPUgetAddressableNumVGPRs) {
             1024u);
 }
 
+TEST(TargetParserTest, testAMDGPUDynamicVGPRAllocGranule) {
+  for (auto Kind :
+       {AMDGPU::GK_GFX1200, AMDGPU::GK_GFX1250, AMDGPU::GK_GFX1310}) {
+    SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+    for (bool IsWave32 : {false, true}) {
+      EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Kind, IsWave32, 16), 16u);
+      EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Kind, IsWave32, 32), 32u);
+    }
+  }
+
+  // Zero selects the static limits, which depend on the wavefront size.
+  EXPECT_EQ(AMDGPU::getVGPRAllocGranule(AMDGPU::GK_GFX1250, false, 0), 8u);
+  EXPECT_EQ(AMDGPU::getVGPRAllocGranule(AMDGPU::GK_GFX1250, true, 0), 16u);
+
+  // gfx90a-family targets always use their fixed allocation granule.
+  for (auto Kind : {AMDGPU::GK_GFX90A, AMDGPU::GK_GFX942, AMDGPU::GK_GFX950}) {
+    SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+    for (bool IsWave32 : {false, true}) {
+      EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Kind, IsWave32, 16), 8u);
+    }
+  }
+
+  EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Triple::AMDGPUSubArch1200, true, 16),
+            16u);
+  EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Triple::AMDGPUSubArch1250, false, 32),
+            32u);
+  EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Triple::AMDGPUSubArch90A, false, 16),
+            8u);
+}
+
 TEST(TargetParserTest, testAMDGPUgetMaxHWAddressableLocalMemorySize) {
   // The addressable cap is a fixed hardware property, independent of how many
   // SIMDs a work-group runs on.



More information about the llvm-branch-commits mailing list