[llvm-branch-commits] [llvm] [AMDGPU] Move `getAddressableNumVGPRs` to TargetParser (PR #226821)
Chinmay Deshpande via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Sun Sep 27 11:30:02 PDT 2026
https://github.com/chinmaydd created https://github.com/llvm/llvm-project/pull/226821
Extend getAddressableNumVGPRs with a dynamic block size and share the eight-block limit through TargetParser. Remove the BaseInfo counterpart and update its subtarget, occupancy, scheduler, and diagnostic callers.
>From 45dde814328f9ecf3cd957719277b7dafe61f698 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sun, 27 Sep 2026 14:15:39 -0400
Subject: [PATCH] [AMDGPU] Move dynamic VGPR addressability to TargetParser
Extend getAddressableNumVGPRs with a dynamic block size and share the
eight-block limit through TargetParser. Remove the BaseInfo counterpart
and update its subtarget, occupancy, scheduler, and diagnostic callers.
Cover dynamic addressability, static limits, and unified register files.
Change-Id: Ie4dae7ba34cce8e8ea50e24f7075242a8a045505
---
.../llvm/TargetParser/AMDGPUTargetParser.h | 15 ++++++---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 13 ++++----
llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp | 3 +-
llvm/lib/Target/AMDGPU/GCNSubtarget.h | 8 +----
.../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 17 ++--------
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 7 -----
llvm/lib/TargetParser/AMDGPUTargetParser.cpp | 11 +++++--
.../llvm-calc-occupancy.cpp | 3 +-
.../TargetParser/TargetParserTest.cpp | 31 +++++++++++++++++++
9 files changed, 61 insertions(+), 47 deletions(-)
diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index d8ad8dcd415c3..e7f758deb2cae 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -196,12 +196,19 @@ LLVM_ABI unsigned getVGPREncodingGranule(Triple::SubArchType SubArch,
LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32);
LLVM_ABI unsigned getTotalNumVGPRs(Triple::SubArchType SubArch, bool IsWave32);
+/// Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
+constexpr unsigned MaxDynamicVGPRBlocks = 8;
+
/// \returns Number of VGPRs a single wave can address. On a target with a
-/// unified register file this covers the AGPRs as well. This does not account
-/// for dynamic VGPR mode, which caps allocation at a fixed number of blocks.
-LLVM_ABI unsigned getAddressableNumVGPRs(GPUKind AK, bool IsWave32);
+/// unified register file this covers the AGPRs as well. A nonzero
+/// \p DynamicVGPRBlockSize selects dynamic VGPR mode, which caps allocation at
+/// \c MaxDynamicVGPRBlocks blocks. On gfx90a-family targets, the unified
+/// register file size is returned regardless of \p DynamicVGPRBlockSize.
+LLVM_ABI unsigned getAddressableNumVGPRs(GPUKind AK, bool IsWave32,
+ unsigned DynamicVGPRBlockSize = 0);
LLVM_ABI unsigned getAddressableNumVGPRs(Triple::SubArchType SubArch,
- bool IsWave32);
+ bool IsWave32,
+ unsigned DynamicVGPRBlockSize = 0);
/// LDS size queries.
///
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index cbe54f4d89832..d038f1a442f64 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1161,13 +1161,12 @@ void AMDGPUAsmPrinter::emitDVgprSymbol(MachineFunction &MF) {
unsigned NumBlocks =
divideCeil(std::max(unsigned(NumVGPRs.getConstant()), 1U), BlockSize);
- if (NumBlocks > AMDGPU::IsaInfo::MaxDynamicVGPRBlocks) {
- OutContext.reportError(
- {}, "DVGPR block count " + Twine(NumBlocks) +
- " exceeds maximum of " +
- Twine(AMDGPU::IsaInfo::MaxDynamicVGPRBlocks) +
- " for __dvgpr$ symbol for '" +
- Twine(CurrentFnSym->getName()) + "'");
+ if (NumBlocks > AMDGPU::MaxDynamicVGPRBlocks) {
+ OutContext.reportError({}, "DVGPR block count " + Twine(NumBlocks) +
+ " exceeds maximum of " +
+ Twine(AMDGPU::MaxDynamicVGPRBlocks) +
+ " for __dvgpr$ symbol for '" +
+ Twine(CurrentFnSym->getName()) + "'");
return;
}
unsigned EncodedNumBlocks = (NumBlocks - 1) << 3;
diff --git a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
index 460c872dc46d6..fd7d643404e39 100644
--- a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
@@ -170,8 +170,7 @@ void GCNSchedStrategy::initialize(ScheduleDAGMI *DAG) {
"VGPRCriticalLimit calculation method.\n");
unsigned DynamicVGPRBlockSize = MFI.getDynamicVGPRBlockSize();
unsigned Granule = ST.getVGPRAllocGranule(DynamicVGPRBlockSize);
- unsigned Addressable =
- AMDGPU::IsaInfo::getAddressableNumVGPRs(ST, DynamicVGPRBlockSize);
+ unsigned Addressable = ST.getAddressableNumVGPRs(DynamicVGPRBlockSize);
unsigned VGPRBudget = alignDown(Addressable / TargetOccupancy, Granule);
VGPRBudget = std::max(VGPRBudget, Granule);
VGPRCriticalLimit = std::min(VGPRBudget, VGPRExcessLimit);
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 437cc9e942d5a..d5660a2c9fb04 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -871,14 +871,8 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
/// \returns Addressable number of VGPRs supported by the subtarget.
unsigned getAddressableNumVGPRs(unsigned DynamicVGPRBlockSize) const {
- // Dynamic VGPR mode is a per-kernel mode, so it is not covered by the
- // TargetParser query.
- if (DynamicVGPRBlockSize != 0) {
- return AMDGPU::IsaInfo::getAddressableNumVGPRs(*this,
- DynamicVGPRBlockSize);
- }
return AMDGPU::getAddressableNumVGPRs(getTargetID().getGPUKind(),
- isWave32());
+ isWave32(), DynamicVGPRBlockSize);
}
/// \returns the minimum number of VGPRs that will prevent achieving more than
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 82a9603915291..63ece72e08879 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1307,19 +1307,6 @@ unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI) {
return 256;
}
-unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
- unsigned DynamicVGPRBlockSize) {
- const auto &Features = STI.getFeatureBits();
- if (Features.test(FeatureGFX90AInsts))
- return 512;
-
- if (DynamicVGPRBlockSize != 0) {
- // On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
- return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
- }
- return getAddressableNumArchVGPRs(STI);
-}
-
unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI,
unsigned NumVGPRs,
unsigned DynamicVGPRBlockSize) {
@@ -1382,7 +1369,7 @@ unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(Kind, IsWave32);
unsigned AddrsableNumVGPRs =
- getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
+ AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, DynamicVGPRBlockSize);
unsigned Granule =
AMDGPU::getVGPRAllocGranule(Kind, IsWave32, DynamicVGPRBlockSize);
unsigned MaxNumVGPRs = alignDown(TotNumVGPRs / WavesPerEU, Granule);
@@ -1416,7 +1403,7 @@ unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
AMDGPU::getVGPRAllocGranule(
Kind, IsWave32, DynamicVGPRBlockSize));
unsigned AddressableNumVGPRs =
- getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
+ AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, DynamicVGPRBlockSize);
return std::min(MaxNumVGPRs, AddressableNumVGPRs);
}
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index a57cfee284738..211771a8dab11 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -237,17 +237,10 @@ unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs);
/// returns the allocation granule for ArchVGPRs.
unsigned getArchVGPRAllocGranule();
-/// Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
-static constexpr unsigned MaxDynamicVGPRBlocks = 8;
-
/// \returns Addressable number of architectural VGPRs for a given subtarget \p
/// STI.
unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI);
-/// \returns Addressable number of VGPRs for given subtarget \p STI.
-unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
- unsigned DynamicVGPRBlockSize);
-
/// \returns Minimum number of VGPRs that meets given number of waves per
/// execution unit requirement for given subtarget \p STI.
unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index bcdf5c356ef98..d82a82bd9699e 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -471,19 +471,24 @@ unsigned AMDGPU::getTotalNumVGPRs(Triple::SubArchType SubArch, bool IsWave32) {
return getTotalNumVGPRs(getGPUKindFromSubArch(SubArch), IsWave32);
}
-unsigned AMDGPU::getAddressableNumVGPRs(GPUKind AK, bool IsWave32) {
+unsigned AMDGPU::getAddressableNumVGPRs(GPUKind AK, bool IsWave32,
+ unsigned DynamicVGPRBlockSize) {
const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
// The unified register file makes the AGPRs addressable as VGPRs.
if (Features.test(FEAT_GFX90A_INSTS))
return 512;
+ if (DynamicVGPRBlockSize != 0)
+ return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
if (Features.test(FEAT_1024_ADDRESSABLE_VGPRS))
return IsWave32 ? 1024 : 512;
return 256;
}
unsigned AMDGPU::getAddressableNumVGPRs(Triple::SubArchType SubArch,
- bool IsWave32) {
- return getAddressableNumVGPRs(getGPUKindFromSubArch(SubArch), IsWave32);
+ bool IsWave32,
+ unsigned DynamicVGPRBlockSize) {
+ return getAddressableNumVGPRs(getGPUKindFromSubArch(SubArch), IsWave32,
+ DynamicVGPRBlockSize);
}
unsigned AMDGPU::getMaxHWAddressableLocalMemorySize(GPUKind AK) {
diff --git a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
index a21165d041bc6..2ef08e4bb70d1 100644
--- a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
+++ b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
@@ -218,8 +218,7 @@ int main(int argc, char **argv) {
unsigned NumWorkGroupSIMDs = ST.getNumWorkGroupSIMDs();
unsigned LocalMemSize = AMDGPU::IsaInfo::getLocalMemorySize(STI);
unsigned AddrLocalMem = AMDGPU::IsaInfo::getAddressableLocalMemorySize(STI);
- unsigned AddrVGPRs =
- AMDGPU::IsaInfo::getAddressableNumVGPRs(STI, DynVGPRBlockSize);
+ unsigned AddrVGPRs = ST.getAddressableNumVGPRs(DynVGPRBlockSize);
unsigned AddrSGPRs = ST.getAddressableNumSGPRs();
unsigned MaxWGSize = AMDGPU::getMaxFlatWorkGroupSize();
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 7c5f156775bd2..1844f929fd849 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -3327,6 +3327,37 @@ TEST(TargetParserTest, testAMDGPUDynamicVGPRAllocGranule) {
8u);
}
+TEST(TargetParserTest, testAMDGPUDynamicVGPRAddressableNum) {
+ for (auto Kind :
+ {AMDGPU::GK_GFX1200, AMDGPU::GK_GFX1250, AMDGPU::GK_GFX1310}) {
+ SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+ for (bool IsWave32 : {false, true}) {
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, 16), 128u);
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, 32), 256u);
+ }
+ }
+
+ // Zero selects the static limits, which depend on the wavefront size.
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(AMDGPU::GK_GFX1250, false, 0), 512u);
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(AMDGPU::GK_GFX1250, true, 0), 1024u);
+
+ // gfx90a-family targets always use their unified register file size.
+ for (auto Kind : {AMDGPU::GK_GFX90A, AMDGPU::GK_GFX942, AMDGPU::GK_GFX950}) {
+ SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+ for (bool IsWave32 : {false, true}) {
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, 16), 512u);
+ }
+ }
+
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Triple::AMDGPUSubArch1200, true, 16),
+ 128u);
+ EXPECT_EQ(
+ AMDGPU::getAddressableNumVGPRs(Triple::AMDGPUSubArch1250, false, 32),
+ 256u);
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Triple::AMDGPUSubArch90A, false, 16),
+ 512u);
+}
+
TEST(TargetParserTest, testAMDGPUgetMaxHWAddressableLocalMemorySize) {
// The addressable cap is a fixed hardware property, independent of how many
// SIMDs a work-group runs on.
More information about the llvm-branch-commits
mailing list