[llvm-branch-commits] [llvm] [AMDGPU] Move dynamic VGPR allocation granules to TargetParser (PR #226818)
Chinmay Deshpande via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Sun Sep 27 11:25:13 PDT 2026
https://github.com/chinmaydd created https://github.com/llvm/llvm-project/pull/226818
Extend getVGPRAllocGranule with a dynamic block size and preserve the fixed gfx90a-family granule. Remove the BaseInfo counterpart and migrate its occupancy, scheduler, frame lowering, and register-block callers.
>From de480cc59fc39991e1fb2eff30b70bffb7aa8227 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sun, 27 Sep 2026 14:13:31 -0400
Subject: [PATCH] [AMDGPU] Move dynamic VGPR allocation granules to
TargetParser
Extend getVGPRAllocGranule with a dynamic block size and preserve the
fixed gfx90a-family granule. Remove the BaseInfo counterpart and migrate
its occupancy, scheduler, frame lowering, and register-block callers.
Cover dynamic allocation, static mode, and gfx90a-family exceptions.
Change-Id: Id441ddcea2d7a33121ddb4139af4c5f52f273ff3
---
.../llvm/TargetParser/AMDGPUTargetParser.h | 11 ++--
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 2 +-
llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp | 3 +-
llvm/lib/Target/AMDGPU/GCNSubtarget.h | 3 +-
llvm/lib/Target/AMDGPU/SIFrameLowering.cpp | 3 +-
.../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 55 ++++++-------------
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 8 ---
llvm/lib/TargetParser/AMDGPUTargetParser.cpp | 12 ++--
.../TargetParser/TargetParserTest.cpp | 30 ++++++++++
9 files changed, 68 insertions(+), 59 deletions(-)
diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index b611e5cbcc4fe..d8ad8dcd415c3 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -174,11 +174,14 @@ LLVM_ABI unsigned getSGPRAllocGranule(Triple::SubArchType SubArch);
/// \returns VGPR allocation granularity for \p AK, in registers. \p IsWave32
/// selects the wavefront size, which is a per-kernel mode rather than a
-/// property of the GPU. This does not account for dynamic VGPR mode, where the
-/// block size chosen by the caller is the granule.
-LLVM_ABI unsigned getVGPRAllocGranule(GPUKind AK, bool IsWave32);
+/// property of the GPU. A nonzero \p DynamicVGPRBlockSize selects dynamic
+/// VGPR mode, where the block size is the granule. On gfx90a-family targets,
+/// the fixed granule is used regardless of \p DynamicVGPRBlockSize.
+LLVM_ABI unsigned getVGPRAllocGranule(GPUKind AK, bool IsWave32,
+ unsigned DynamicVGPRBlockSize = 0);
LLVM_ABI unsigned getVGPRAllocGranule(Triple::SubArchType SubArch,
- bool IsWave32);
+ bool IsWave32,
+ unsigned DynamicVGPRBlockSize = 0);
/// \returns VGPR encoding granularity for \p AK, in registers. \p IsWave32
/// selects the wavefront size encoded in the kernel descriptor. This can
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 89105a78c4e80..cbe54f4d89832 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -438,7 +438,7 @@ const AMDGPUMCExpr *createOccupancy(unsigned InitOcc, const MCExpr *NumSGPRs,
unsigned DynamicVGPRBlockSize,
const GCNSubtarget &STM, MCContext &Ctx) {
unsigned MaxWaves = STM.getMaxWavesPerEU();
- unsigned Granule = IsaInfo::getVGPRAllocGranule(STM, DynamicVGPRBlockSize);
+ unsigned Granule = STM.getVGPRAllocGranule(DynamicVGPRBlockSize);
unsigned TargetTotalNumVGPRs = STM.getTotalNumVGPRs();
// Bake the per-function SGPR budget into the operands so the late-evaluated
diff --git a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
index 602489ab3a5c1..460c872dc46d6 100644
--- a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
@@ -169,8 +169,7 @@ void GCNSchedStrategy::initialize(ScheduleDAGMI *DAG) {
LLVM_DEBUG(dbgs() << "Region is known to spill, use alternative "
"VGPRCriticalLimit calculation method.\n");
unsigned DynamicVGPRBlockSize = MFI.getDynamicVGPRBlockSize();
- unsigned Granule =
- AMDGPU::IsaInfo::getVGPRAllocGranule(ST, DynamicVGPRBlockSize);
+ unsigned Granule = ST.getVGPRAllocGranule(DynamicVGPRBlockSize);
unsigned Addressable =
AMDGPU::IsaInfo::getAddressableNumVGPRs(ST, DynamicVGPRBlockSize);
unsigned VGPRBudget = alignDown(Addressable / TargetOccupancy, Granule);
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index c344367b460f4..437cc9e942d5a 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -848,7 +848,8 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
/// \returns VGPR allocation granularity supported by the subtarget.
unsigned getVGPRAllocGranule(unsigned DynamicVGPRBlockSize) const {
- return AMDGPU::IsaInfo::getVGPRAllocGranule(*this, DynamicVGPRBlockSize);
+ return AMDGPU::getVGPRAllocGranule(getTargetID().getGPUKind(), isWave32(),
+ DynamicVGPRBlockSize);
}
/// \returns VGPR encoding granularity supported by the subtarget.
diff --git a/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp b/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
index 364d8f07b6df2..81a371352afcc 100644
--- a/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
@@ -889,8 +889,7 @@ void SIFrameLowering::emitEntryFunctionPrologue(MachineFunction &MF,
assert(FPReg != AMDGPU::FP_REG);
unsigned VGPRSize = llvm::alignTo(
(ST.getAddressableNumVGPRs(MFI->getDynamicVGPRBlockSize()) -
- AMDGPU::IsaInfo::getVGPRAllocGranule(ST,
- MFI->getDynamicVGPRBlockSize())) *
+ ST.getVGPRAllocGranule(MFI->getDynamicVGPRBlockSize())) *
4,
FrameInfo.getMaxAlign());
MFI->setScratchReservedForDynamicVGPRs(VGPRSize);
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index baaef6293b9b0..82a9603915291 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1298,28 +1298,6 @@ unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs) {
1;
}
-unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI,
- unsigned DynamicVGPRBlockSize,
- std::optional<bool> EnableWavefrontSize32) {
- if (STI.getFeatureBits().test(FeatureGFX90AInsts))
- return 8;
-
- if (DynamicVGPRBlockSize != 0)
- return DynamicVGPRBlockSize;
-
- bool IsWave32 = EnableWavefrontSize32
- ? *EnableWavefrontSize32
- : STI.getFeatureBits().test(FeatureWavefrontSize32);
-
- if (STI.getFeatureBits().test(Feature1536VGPRs))
- return IsWave32 ? 24 : 12;
-
- if (hasGFX10_3Insts(STI))
- return IsWave32 ? 16 : 8;
-
- return IsWave32 ? 8 : 4;
-}
-
unsigned getArchVGPRAllocGranule() { return 4; }
unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI) {
@@ -1337,8 +1315,7 @@ unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
if (DynamicVGPRBlockSize != 0) {
// On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
- return MaxDynamicVGPRBlocks *
- getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
+ return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
}
return getAddressableNumArchVGPRs(STI);
}
@@ -1349,7 +1326,8 @@ unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI,
GPUKind Kind = parseArchAMDGCN(STI.getCPU());
bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
return getNumWavesPerEUWithNumVGPRs(
- NumVGPRs, getVGPRAllocGranule(STI, DynamicVGPRBlockSize),
+ NumVGPRs,
+ AMDGPU::getVGPRAllocGranule(Kind, IsWave32, DynamicVGPRBlockSize),
getMaxWavesPerEU(Kind), AMDGPU::getTotalNumVGPRs(Kind, IsWave32));
}
@@ -1401,11 +1379,12 @@ unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
if (WavesPerEU >= MaxWavesPerEU)
return 0;
- unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(
- Kind, STI.getFeatureBits().test(FeatureWavefrontSize32));
+ bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
+ unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(Kind, IsWave32);
unsigned AddrsableNumVGPRs =
getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
- unsigned Granule = getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
+ unsigned Granule =
+ AMDGPU::getVGPRAllocGranule(Kind, IsWave32, DynamicVGPRBlockSize);
unsigned MaxNumVGPRs = alignDown(TotNumVGPRs / WavesPerEU, Granule);
if (MaxNumVGPRs == alignDown(TotNumVGPRs / MaxWavesPerEU, Granule))
@@ -1425,17 +1404,17 @@ unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
unsigned DynamicVGPRBlockSize) {
assert(WavesPerEU != 0);
- unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(
- parseArchAMDGCN(STI.getCPU()),
- STI.getFeatureBits().test(FeatureWavefrontSize32));
+ GPUKind Kind = parseArchAMDGCN(STI.getCPU());
+ bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
+ unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(Kind, IsWave32);
// In dynamic VGPR mode, WavesPerEU does not imply a VGPR limit.
bool DynamicVGPREnabled = (DynamicVGPRBlockSize != 0);
unsigned MaxNumVGPRs =
- DynamicVGPREnabled
- ? TotNumVGPRs
- : alignDown(TotNumVGPRs / WavesPerEU,
- getVGPRAllocGranule(STI, DynamicVGPRBlockSize));
+ DynamicVGPREnabled ? TotNumVGPRs
+ : alignDown(TotNumVGPRs / WavesPerEU,
+ AMDGPU::getVGPRAllocGranule(
+ Kind, IsWave32, DynamicVGPRBlockSize));
unsigned AddressableNumVGPRs =
getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
return std::min(MaxNumVGPRs, AddressableNumVGPRs);
@@ -1455,9 +1434,11 @@ unsigned getAllocatedNumVGPRBlocks(const MCSubtargetInfo &STI,
unsigned NumVGPRs,
unsigned DynamicVGPRBlockSize,
std::optional<bool> EnableWavefrontSize32) {
+ bool IsWave32 = EnableWavefrontSize32.value_or(
+ STI.getFeatureBits().test(FeatureWavefrontSize32));
return getGranulatedNumRegisterBlocks(
- NumVGPRs,
- getVGPRAllocGranule(STI, DynamicVGPRBlockSize, EnableWavefrontSize32));
+ NumVGPRs, AMDGPU::getVGPRAllocGranule(parseArchAMDGCN(STI.getCPU()),
+ IsWave32, DynamicVGPRBlockSize));
}
} // end namespace IsaInfo
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index 564d77b0ad938..a57cfee284738 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -233,14 +233,6 @@ unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed,
/// register counts.
unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs);
-/// \returns VGPR allocation granularity for given subtarget \p STI.
-///
-/// For subtargets which support it, \p EnableWavefrontSize32 should match
-/// the ENABLE_WAVEFRONT_SIZE32 kernel descriptor field.
-unsigned
-getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize,
- std::optional<bool> EnableWavefrontSize32 = std::nullopt);
-
/// For subtargets with a unified VGPR file and mixed ArchVGPR/AGPR usage,
/// returns the allocation granule for ArchVGPRs.
unsigned getArchVGPRAllocGranule();
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index 359b65363f2d1..bcdf5c356ef98 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -422,10 +422,13 @@ unsigned AMDGPU::getSGPRAllocGranule(Triple::SubArchType SubArch) {
return 8;
}
-unsigned AMDGPU::getVGPRAllocGranule(GPUKind AK, bool IsWave32) {
+unsigned AMDGPU::getVGPRAllocGranule(GPUKind AK, bool IsWave32,
+ unsigned DynamicVGPRBlockSize) {
const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
if (Features.test(FEAT_GFX90A_INSTS))
return 8;
+ if (DynamicVGPRBlockSize != 0)
+ return DynamicVGPRBlockSize;
if (Features.test(FEAT_1536_PHYSICAL_VGPRS))
return IsWave32 ? 24 : 12;
if (Features.test(FEAT_GFX10_3_INSTS))
@@ -433,9 +436,10 @@ unsigned AMDGPU::getVGPRAllocGranule(GPUKind AK, bool IsWave32) {
return IsWave32 ? 8 : 4;
}
-unsigned AMDGPU::getVGPRAllocGranule(Triple::SubArchType SubArch,
- bool IsWave32) {
- return getVGPRAllocGranule(getGPUKindFromSubArch(SubArch), IsWave32);
+unsigned AMDGPU::getVGPRAllocGranule(Triple::SubArchType SubArch, bool IsWave32,
+ unsigned DynamicVGPRBlockSize) {
+ return getVGPRAllocGranule(getGPUKindFromSubArch(SubArch), IsWave32,
+ DynamicVGPRBlockSize);
}
unsigned AMDGPU::getVGPREncodingGranule(GPUKind AK, bool IsWave32) {
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 2c963026a6a45..7c5f156775bd2 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -3297,6 +3297,36 @@ TEST(TargetParserTest, testAMDGPUgetAddressableNumVGPRs) {
1024u);
}
+TEST(TargetParserTest, testAMDGPUDynamicVGPRAllocGranule) {
+ for (auto Kind :
+ {AMDGPU::GK_GFX1200, AMDGPU::GK_GFX1250, AMDGPU::GK_GFX1310}) {
+ SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+ for (bool IsWave32 : {false, true}) {
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Kind, IsWave32, 16), 16u);
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Kind, IsWave32, 32), 32u);
+ }
+ }
+
+ // Zero selects the static limits, which depend on the wavefront size.
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(AMDGPU::GK_GFX1250, false, 0), 8u);
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(AMDGPU::GK_GFX1250, true, 0), 16u);
+
+ // gfx90a-family targets always use their fixed allocation granule.
+ for (auto Kind : {AMDGPU::GK_GFX90A, AMDGPU::GK_GFX942, AMDGPU::GK_GFX950}) {
+ SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+ for (bool IsWave32 : {false, true}) {
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Kind, IsWave32, 16), 8u);
+ }
+ }
+
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Triple::AMDGPUSubArch1200, true, 16),
+ 16u);
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Triple::AMDGPUSubArch1250, false, 32),
+ 32u);
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Triple::AMDGPUSubArch90A, false, 16),
+ 8u);
+}
+
TEST(TargetParserTest, testAMDGPUgetMaxHWAddressableLocalMemorySize) {
// The addressable cap is a fixed hardware property, independent of how many
// SIMDs a work-group runs on.
More information about the llvm-branch-commits
mailing list