[llvm] [AMDGPU] Remove BaseInfo duplicates of TargetParser APIs (PR #226744)
Chinmay Deshpande via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 27 11:31:02 PDT 2026
https://github.com/chinmaydd updated https://github.com/llvm/llvm-project/pull/226744
>From e975ae3962a15484dba8df32af8181782f39451b Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sun, 27 Sep 2026 14:11:21 -0400
Subject: [PATCH 1/4] [AMDGPU] Move getVGPREncodingGranule to TargetParser
Add GPUKind and SubArch encoding-granule queries with an explicit wave size.
Remove the BaseInfo counterpart and update codegen, assembler, and
disassembler callers, preserving kernel descriptor wave-size overrides.
Cover encoding granules across GPU generations and both wave sizes.
Change-Id: I1982fef776e5a0b7c613c274e544ba3e3f86f136
---
.../llvm/TargetParser/AMDGPUTargetParser.h | 8 ++++++
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 2 +-
.../AMDGPU/AsmParser/AMDGPUAsmParser.cpp | 7 +++--
.../Disassembler/AMDGPUDisassembler.cpp | 4 ++-
llvm/lib/Target/AMDGPU/GCNSubtarget.h | 3 +-
.../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 20 +++----------
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 8 ------
llvm/lib/TargetParser/AMDGPUTargetParser.cpp | 14 ++++++++++
.../TargetParser/TargetParserTest.cpp | 28 +++++++++++++++++++
9 files changed, 64 insertions(+), 30 deletions(-)
diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index 042542ced349d..b611e5cbcc4fe 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -180,6 +180,14 @@ LLVM_ABI unsigned getVGPRAllocGranule(GPUKind AK, bool IsWave32);
LLVM_ABI unsigned getVGPRAllocGranule(Triple::SubArchType SubArch,
bool IsWave32);
+/// \returns VGPR encoding granularity for \p AK, in registers. \p IsWave32
+/// selects the wavefront size encoded in the kernel descriptor. This can
+/// differ from the allocation granularity and is independent of dynamic
+/// VGPR mode.
+LLVM_ABI unsigned getVGPREncodingGranule(GPUKind AK, bool IsWave32);
+LLVM_ABI unsigned getVGPREncodingGranule(Triple::SubArchType SubArch,
+ bool IsWave32);
+
/// \returns Number of physical VGPRs, i.e. the size of the register file a
/// work-group's waves share. \p IsWave32 selects the wavefront size.
LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32);
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index c18b840c36791..89105a78c4e80 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1431,7 +1431,7 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
IsaInfo::getSGPREncodingGranule(STM));
}
ProgInfo.VGPRBlocks = GetNumGPRBlocks(ProgInfo.NumVGPRsForWavesPerEU,
- IsaInfo::getVGPREncodingGranule(STM));
+ STM.getVGPREncodingGranule());
const SIModeRegisterDefaults Mode = MFI->getMode();
diff --git a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
index 812acc8f8be85..3a7213e52d5f5 100644
--- a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
+++ b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
@@ -6193,9 +6193,10 @@ bool AMDGPUAsmParser::calculateGPRBlocks(
return SubGPR;
};
- VGPRBlocks = GetNumGPRBlocks(
- NextFreeVGPR,
- IsaInfo::getVGPREncodingGranule(getSTI(), EnableWavefrontSize32));
+ bool IsWave32 = EnableWavefrontSize32.value_or(
+ getSTI().getFeatureBits().test(FeatureWavefrontSize32));
+ VGPRBlocks = GetNumGPRBlocks(NextFreeVGPR,
+ AMDGPU::getVGPREncodingGranule(Gfx, IsWave32));
SGPRBlocks =
GetNumGPRBlocks(NumSGPRs, IsaInfo::getSGPREncodingGranule(getSTI()));
diff --git a/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp b/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
index 3b200d412eb0d..8d930b3761859 100644
--- a/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
+++ b/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
@@ -2594,9 +2594,11 @@ Expected<bool> AMDGPUDisassembler::decodeCOMPUTE_PGM_RSRC1(
uint32_t GranulatedWorkitemVGPRCount =
GET_FIELD(COMPUTE_PGM_RSRC1_GRANULATED_WORKITEM_VGPR_COUNT);
+ bool IsWave32 = EnableWavefrontSize32.value_or(
+ STI.getFeatureBits().test(AMDGPU::FeatureWavefrontSize32));
uint32_t NextFreeVGPR =
(GranulatedWorkitemVGPRCount + 1) *
- AMDGPU::IsaInfo::getVGPREncodingGranule(STI, EnableWavefrontSize32);
+ AMDGPU::getVGPREncodingGranule(TargetID.getGPUKind(), IsWave32);
KdStream << Indent << ".amdhsa_next_free_vgpr " << NextFreeVGPR << '\n';
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 2e09f1f1fda39..c344367b460f4 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -853,7 +853,8 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
/// \returns VGPR encoding granularity supported by the subtarget.
unsigned getVGPREncodingGranule() const {
- return AMDGPU::IsaInfo::getVGPREncodingGranule(*this);
+ return AMDGPU::getVGPREncodingGranule(getTargetID().getGPUKind(),
+ isWave32());
}
/// \returns Total number of VGPRs supported by the subtarget.
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index e45359719d6c9..baaef6293b9b0 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1320,21 +1320,6 @@ unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI,
return IsWave32 ? 8 : 4;
}
-unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI,
- std::optional<bool> EnableWavefrontSize32) {
- if (STI.getFeatureBits().test(FeatureGFX90AInsts))
- return 8;
-
- bool IsWave32 = EnableWavefrontSize32
- ? *EnableWavefrontSize32
- : STI.getFeatureBits().test(FeatureWavefrontSize32);
-
- if (STI.getFeatureBits().test(Feature1024AddressableVGPRs))
- return IsWave32 ? 16 : 8;
-
- return IsWave32 ? 8 : 4;
-}
-
unsigned getArchVGPRAllocGranule() { return 4; }
unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI) {
@@ -1458,8 +1443,11 @@ unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
unsigned getEncodedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs,
std::optional<bool> EnableWavefrontSize32) {
+ bool IsWave32 = EnableWavefrontSize32.value_or(
+ STI.getFeatureBits().test(FeatureWavefrontSize32));
return getGranulatedNumRegisterBlocks(
- NumVGPRs, getVGPREncodingGranule(STI, EnableWavefrontSize32)) -
+ NumVGPRs, AMDGPU::getVGPREncodingGranule(
+ parseArchAMDGCN(STI.getCPU()), IsWave32)) -
1;
}
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index bce059a0c18a7..564d77b0ad938 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -241,14 +241,6 @@ unsigned
getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize,
std::optional<bool> EnableWavefrontSize32 = std::nullopt);
-/// \returns VGPR encoding granularity for given subtarget \p STI.
-///
-/// For subtargets which support it, \p EnableWavefrontSize32 should match
-/// the ENABLE_WAVEFRONT_SIZE32 kernel descriptor field.
-unsigned getVGPREncodingGranule(
- const MCSubtargetInfo &STI,
- std::optional<bool> EnableWavefrontSize32 = std::nullopt);
-
/// For subtargets with a unified VGPR file and mixed ArchVGPR/AGPR usage,
/// returns the allocation granule for ArchVGPRs.
unsigned getArchVGPRAllocGranule();
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index 756d665353c45..359b65363f2d1 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -438,6 +438,20 @@ unsigned AMDGPU::getVGPRAllocGranule(Triple::SubArchType SubArch,
return getVGPRAllocGranule(getGPUKindFromSubArch(SubArch), IsWave32);
}
+unsigned AMDGPU::getVGPREncodingGranule(GPUKind AK, bool IsWave32) {
+ const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
+ if (Features.test(FEAT_GFX90A_INSTS))
+ return 8;
+ if (Features.test(FEAT_1024_ADDRESSABLE_VGPRS))
+ return IsWave32 ? 16 : 8;
+ return IsWave32 ? 8 : 4;
+}
+
+unsigned AMDGPU::getVGPREncodingGranule(Triple::SubArchType SubArch,
+ bool IsWave32) {
+ return getVGPREncodingGranule(getGPUKindFromSubArch(SubArch), IsWave32);
+}
+
unsigned AMDGPU::getTotalNumVGPRs(GPUKind AK, bool IsWave32) {
const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
if (Features.test(FEAT_GFX90A_INSTS))
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 62d6f3a4c955b..2c963026a6a45 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -3226,6 +3226,34 @@ TEST(TargetParserTest, testAMDGPUgetVGPRAllocGranule) {
EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Triple::AMDGPUSubArch1100, true), 24u);
}
+TEST(TargetParserTest, testAMDGPUgetVGPREncodingGranule) {
+ for (auto Kind :
+ {AMDGPU::GK_NONE, AMDGPU::GK_GFX600, AMDGPU::GK_GFX1030,
+ AMDGPU::GK_GFX1100, AMDGPU::GK_GFX1200, AMDGPU::GK_GFX1310}) {
+ SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+ EXPECT_EQ(AMDGPU::getVGPREncodingGranule(Kind, false), 4u);
+ EXPECT_EQ(AMDGPU::getVGPREncodingGranule(Kind, true), 8u);
+ }
+ for (auto Kind : {AMDGPU::GK_GFX90A, AMDGPU::GK_GFX942, AMDGPU::GK_GFX950}) {
+ SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+ EXPECT_EQ(AMDGPU::getVGPREncodingGranule(Kind, false), 8u);
+ EXPECT_EQ(AMDGPU::getVGPREncodingGranule(Kind, true), 8u);
+ }
+ EXPECT_EQ(AMDGPU::getVGPREncodingGranule(AMDGPU::GK_GFX1250, false), 8u);
+ EXPECT_EQ(AMDGPU::getVGPREncodingGranule(AMDGPU::GK_GFX1250, true), 16u);
+
+ // Encoding granules can be smaller than allocation granules.
+ EXPECT_EQ(AMDGPU::getVGPREncodingGranule(Triple::AMDGPUSubArch1030, true),
+ 8u);
+ EXPECT_EQ(AMDGPU::getVGPREncodingGranule(Triple::AMDGPUSubArch1100, false),
+ 4u);
+ EXPECT_EQ(AMDGPU::getVGPREncodingGranule(Triple::AMDGPUSubArch90A, false),
+ 8u);
+ EXPECT_EQ(AMDGPU::getVGPREncodingGranule(Triple::AMDGPUSubArch1250, true),
+ 16u);
+ EXPECT_EQ(AMDGPU::getVGPREncodingGranule(Triple::NoSubArch, false), 4u);
+}
+
TEST(TargetParserTest, testAMDGPUgetTotalNumVGPRs) {
// Pre-gfx10 the file is 256 registers, and gfx90a doubles it by unifying the
// AGPRs. From gfx10 on it is split between the waves of a wave64 kernel.
>From de480cc59fc39991e1fb2eff30b70bffb7aa8227 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sun, 27 Sep 2026 14:13:31 -0400
Subject: [PATCH 2/4] [AMDGPU] Move dynamic VGPR allocation granules to
TargetParser
Extend getVGPRAllocGranule with a dynamic block size and preserve the
fixed gfx90a-family granule. Remove the BaseInfo counterpart and migrate
its occupancy, scheduler, frame lowering, and register-block callers.
Cover dynamic allocation, static mode, and gfx90a-family exceptions.
Change-Id: Id441ddcea2d7a33121ddb4139af4c5f52f273ff3
---
.../llvm/TargetParser/AMDGPUTargetParser.h | 11 ++--
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 2 +-
llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp | 3 +-
llvm/lib/Target/AMDGPU/GCNSubtarget.h | 3 +-
llvm/lib/Target/AMDGPU/SIFrameLowering.cpp | 3 +-
.../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 55 ++++++-------------
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 8 ---
llvm/lib/TargetParser/AMDGPUTargetParser.cpp | 12 ++--
.../TargetParser/TargetParserTest.cpp | 30 ++++++++++
9 files changed, 68 insertions(+), 59 deletions(-)
diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index b611e5cbcc4fe..d8ad8dcd415c3 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -174,11 +174,14 @@ LLVM_ABI unsigned getSGPRAllocGranule(Triple::SubArchType SubArch);
/// \returns VGPR allocation granularity for \p AK, in registers. \p IsWave32
/// selects the wavefront size, which is a per-kernel mode rather than a
-/// property of the GPU. This does not account for dynamic VGPR mode, where the
-/// block size chosen by the caller is the granule.
-LLVM_ABI unsigned getVGPRAllocGranule(GPUKind AK, bool IsWave32);
+/// property of the GPU. A nonzero \p DynamicVGPRBlockSize selects dynamic
+/// VGPR mode, where the block size is the granule. On gfx90a-family targets,
+/// the fixed granule is used regardless of \p DynamicVGPRBlockSize.
+LLVM_ABI unsigned getVGPRAllocGranule(GPUKind AK, bool IsWave32,
+ unsigned DynamicVGPRBlockSize = 0);
LLVM_ABI unsigned getVGPRAllocGranule(Triple::SubArchType SubArch,
- bool IsWave32);
+ bool IsWave32,
+ unsigned DynamicVGPRBlockSize = 0);
/// \returns VGPR encoding granularity for \p AK, in registers. \p IsWave32
/// selects the wavefront size encoded in the kernel descriptor. This can
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 89105a78c4e80..cbe54f4d89832 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -438,7 +438,7 @@ const AMDGPUMCExpr *createOccupancy(unsigned InitOcc, const MCExpr *NumSGPRs,
unsigned DynamicVGPRBlockSize,
const GCNSubtarget &STM, MCContext &Ctx) {
unsigned MaxWaves = STM.getMaxWavesPerEU();
- unsigned Granule = IsaInfo::getVGPRAllocGranule(STM, DynamicVGPRBlockSize);
+ unsigned Granule = STM.getVGPRAllocGranule(DynamicVGPRBlockSize);
unsigned TargetTotalNumVGPRs = STM.getTotalNumVGPRs();
// Bake the per-function SGPR budget into the operands so the late-evaluated
diff --git a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
index 602489ab3a5c1..460c872dc46d6 100644
--- a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
@@ -169,8 +169,7 @@ void GCNSchedStrategy::initialize(ScheduleDAGMI *DAG) {
LLVM_DEBUG(dbgs() << "Region is known to spill, use alternative "
"VGPRCriticalLimit calculation method.\n");
unsigned DynamicVGPRBlockSize = MFI.getDynamicVGPRBlockSize();
- unsigned Granule =
- AMDGPU::IsaInfo::getVGPRAllocGranule(ST, DynamicVGPRBlockSize);
+ unsigned Granule = ST.getVGPRAllocGranule(DynamicVGPRBlockSize);
unsigned Addressable =
AMDGPU::IsaInfo::getAddressableNumVGPRs(ST, DynamicVGPRBlockSize);
unsigned VGPRBudget = alignDown(Addressable / TargetOccupancy, Granule);
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index c344367b460f4..437cc9e942d5a 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -848,7 +848,8 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
/// \returns VGPR allocation granularity supported by the subtarget.
unsigned getVGPRAllocGranule(unsigned DynamicVGPRBlockSize) const {
- return AMDGPU::IsaInfo::getVGPRAllocGranule(*this, DynamicVGPRBlockSize);
+ return AMDGPU::getVGPRAllocGranule(getTargetID().getGPUKind(), isWave32(),
+ DynamicVGPRBlockSize);
}
/// \returns VGPR encoding granularity supported by the subtarget.
diff --git a/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp b/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
index 364d8f07b6df2..81a371352afcc 100644
--- a/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
@@ -889,8 +889,7 @@ void SIFrameLowering::emitEntryFunctionPrologue(MachineFunction &MF,
assert(FPReg != AMDGPU::FP_REG);
unsigned VGPRSize = llvm::alignTo(
(ST.getAddressableNumVGPRs(MFI->getDynamicVGPRBlockSize()) -
- AMDGPU::IsaInfo::getVGPRAllocGranule(ST,
- MFI->getDynamicVGPRBlockSize())) *
+ ST.getVGPRAllocGranule(MFI->getDynamicVGPRBlockSize())) *
4,
FrameInfo.getMaxAlign());
MFI->setScratchReservedForDynamicVGPRs(VGPRSize);
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index baaef6293b9b0..82a9603915291 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1298,28 +1298,6 @@ unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs) {
1;
}
-unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI,
- unsigned DynamicVGPRBlockSize,
- std::optional<bool> EnableWavefrontSize32) {
- if (STI.getFeatureBits().test(FeatureGFX90AInsts))
- return 8;
-
- if (DynamicVGPRBlockSize != 0)
- return DynamicVGPRBlockSize;
-
- bool IsWave32 = EnableWavefrontSize32
- ? *EnableWavefrontSize32
- : STI.getFeatureBits().test(FeatureWavefrontSize32);
-
- if (STI.getFeatureBits().test(Feature1536VGPRs))
- return IsWave32 ? 24 : 12;
-
- if (hasGFX10_3Insts(STI))
- return IsWave32 ? 16 : 8;
-
- return IsWave32 ? 8 : 4;
-}
-
unsigned getArchVGPRAllocGranule() { return 4; }
unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI) {
@@ -1337,8 +1315,7 @@ unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
if (DynamicVGPRBlockSize != 0) {
// On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
- return MaxDynamicVGPRBlocks *
- getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
+ return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
}
return getAddressableNumArchVGPRs(STI);
}
@@ -1349,7 +1326,8 @@ unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI,
GPUKind Kind = parseArchAMDGCN(STI.getCPU());
bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
return getNumWavesPerEUWithNumVGPRs(
- NumVGPRs, getVGPRAllocGranule(STI, DynamicVGPRBlockSize),
+ NumVGPRs,
+ AMDGPU::getVGPRAllocGranule(Kind, IsWave32, DynamicVGPRBlockSize),
getMaxWavesPerEU(Kind), AMDGPU::getTotalNumVGPRs(Kind, IsWave32));
}
@@ -1401,11 +1379,12 @@ unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
if (WavesPerEU >= MaxWavesPerEU)
return 0;
- unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(
- Kind, STI.getFeatureBits().test(FeatureWavefrontSize32));
+ bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
+ unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(Kind, IsWave32);
unsigned AddrsableNumVGPRs =
getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
- unsigned Granule = getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
+ unsigned Granule =
+ AMDGPU::getVGPRAllocGranule(Kind, IsWave32, DynamicVGPRBlockSize);
unsigned MaxNumVGPRs = alignDown(TotNumVGPRs / WavesPerEU, Granule);
if (MaxNumVGPRs == alignDown(TotNumVGPRs / MaxWavesPerEU, Granule))
@@ -1425,17 +1404,17 @@ unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
unsigned DynamicVGPRBlockSize) {
assert(WavesPerEU != 0);
- unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(
- parseArchAMDGCN(STI.getCPU()),
- STI.getFeatureBits().test(FeatureWavefrontSize32));
+ GPUKind Kind = parseArchAMDGCN(STI.getCPU());
+ bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
+ unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(Kind, IsWave32);
// In dynamic VGPR mode, WavesPerEU does not imply a VGPR limit.
bool DynamicVGPREnabled = (DynamicVGPRBlockSize != 0);
unsigned MaxNumVGPRs =
- DynamicVGPREnabled
- ? TotNumVGPRs
- : alignDown(TotNumVGPRs / WavesPerEU,
- getVGPRAllocGranule(STI, DynamicVGPRBlockSize));
+ DynamicVGPREnabled ? TotNumVGPRs
+ : alignDown(TotNumVGPRs / WavesPerEU,
+ AMDGPU::getVGPRAllocGranule(
+ Kind, IsWave32, DynamicVGPRBlockSize));
unsigned AddressableNumVGPRs =
getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
return std::min(MaxNumVGPRs, AddressableNumVGPRs);
@@ -1455,9 +1434,11 @@ unsigned getAllocatedNumVGPRBlocks(const MCSubtargetInfo &STI,
unsigned NumVGPRs,
unsigned DynamicVGPRBlockSize,
std::optional<bool> EnableWavefrontSize32) {
+ bool IsWave32 = EnableWavefrontSize32.value_or(
+ STI.getFeatureBits().test(FeatureWavefrontSize32));
return getGranulatedNumRegisterBlocks(
- NumVGPRs,
- getVGPRAllocGranule(STI, DynamicVGPRBlockSize, EnableWavefrontSize32));
+ NumVGPRs, AMDGPU::getVGPRAllocGranule(parseArchAMDGCN(STI.getCPU()),
+ IsWave32, DynamicVGPRBlockSize));
}
} // end namespace IsaInfo
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index 564d77b0ad938..a57cfee284738 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -233,14 +233,6 @@ unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed,
/// register counts.
unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs);
-/// \returns VGPR allocation granularity for given subtarget \p STI.
-///
-/// For subtargets which support it, \p EnableWavefrontSize32 should match
-/// the ENABLE_WAVEFRONT_SIZE32 kernel descriptor field.
-unsigned
-getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize,
- std::optional<bool> EnableWavefrontSize32 = std::nullopt);
-
/// For subtargets with a unified VGPR file and mixed ArchVGPR/AGPR usage,
/// returns the allocation granule for ArchVGPRs.
unsigned getArchVGPRAllocGranule();
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index 359b65363f2d1..bcdf5c356ef98 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -422,10 +422,13 @@ unsigned AMDGPU::getSGPRAllocGranule(Triple::SubArchType SubArch) {
return 8;
}
-unsigned AMDGPU::getVGPRAllocGranule(GPUKind AK, bool IsWave32) {
+unsigned AMDGPU::getVGPRAllocGranule(GPUKind AK, bool IsWave32,
+ unsigned DynamicVGPRBlockSize) {
const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
if (Features.test(FEAT_GFX90A_INSTS))
return 8;
+ if (DynamicVGPRBlockSize != 0)
+ return DynamicVGPRBlockSize;
if (Features.test(FEAT_1536_PHYSICAL_VGPRS))
return IsWave32 ? 24 : 12;
if (Features.test(FEAT_GFX10_3_INSTS))
@@ -433,9 +436,10 @@ unsigned AMDGPU::getVGPRAllocGranule(GPUKind AK, bool IsWave32) {
return IsWave32 ? 8 : 4;
}
-unsigned AMDGPU::getVGPRAllocGranule(Triple::SubArchType SubArch,
- bool IsWave32) {
- return getVGPRAllocGranule(getGPUKindFromSubArch(SubArch), IsWave32);
+unsigned AMDGPU::getVGPRAllocGranule(Triple::SubArchType SubArch, bool IsWave32,
+ unsigned DynamicVGPRBlockSize) {
+ return getVGPRAllocGranule(getGPUKindFromSubArch(SubArch), IsWave32,
+ DynamicVGPRBlockSize);
}
unsigned AMDGPU::getVGPREncodingGranule(GPUKind AK, bool IsWave32) {
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 2c963026a6a45..7c5f156775bd2 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -3297,6 +3297,36 @@ TEST(TargetParserTest, testAMDGPUgetAddressableNumVGPRs) {
1024u);
}
+TEST(TargetParserTest, testAMDGPUDynamicVGPRAllocGranule) {
+ for (auto Kind :
+ {AMDGPU::GK_GFX1200, AMDGPU::GK_GFX1250, AMDGPU::GK_GFX1310}) {
+ SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+ for (bool IsWave32 : {false, true}) {
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Kind, IsWave32, 16), 16u);
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Kind, IsWave32, 32), 32u);
+ }
+ }
+
+ // Zero selects the static limits, which depend on the wavefront size.
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(AMDGPU::GK_GFX1250, false, 0), 8u);
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(AMDGPU::GK_GFX1250, true, 0), 16u);
+
+ // gfx90a-family targets always use their fixed allocation granule.
+ for (auto Kind : {AMDGPU::GK_GFX90A, AMDGPU::GK_GFX942, AMDGPU::GK_GFX950}) {
+ SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+ for (bool IsWave32 : {false, true}) {
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Kind, IsWave32, 16), 8u);
+ }
+ }
+
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Triple::AMDGPUSubArch1200, true, 16),
+ 16u);
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Triple::AMDGPUSubArch1250, false, 32),
+ 32u);
+ EXPECT_EQ(AMDGPU::getVGPRAllocGranule(Triple::AMDGPUSubArch90A, false, 16),
+ 8u);
+}
+
TEST(TargetParserTest, testAMDGPUgetMaxHWAddressableLocalMemorySize) {
// The addressable cap is a fixed hardware property, independent of how many
// SIMDs a work-group runs on.
>From 45dde814328f9ecf3cd957719277b7dafe61f698 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sun, 27 Sep 2026 14:15:39 -0400
Subject: [PATCH 3/4] [AMDGPU] Move dynamic VGPR addressability to TargetParser
Extend getAddressableNumVGPRs with a dynamic block size and share the
eight-block limit through TargetParser. Remove the BaseInfo counterpart
and update its subtarget, occupancy, scheduler, and diagnostic callers.
Cover dynamic addressability, static limits, and unified register files.
Change-Id: Ie4dae7ba34cce8e8ea50e24f7075242a8a045505
---
.../llvm/TargetParser/AMDGPUTargetParser.h | 15 ++++++---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 13 ++++----
llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp | 3 +-
llvm/lib/Target/AMDGPU/GCNSubtarget.h | 8 +----
.../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 17 ++--------
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 7 -----
llvm/lib/TargetParser/AMDGPUTargetParser.cpp | 11 +++++--
.../llvm-calc-occupancy.cpp | 3 +-
.../TargetParser/TargetParserTest.cpp | 31 +++++++++++++++++++
9 files changed, 61 insertions(+), 47 deletions(-)
diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index d8ad8dcd415c3..e7f758deb2cae 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -196,12 +196,19 @@ LLVM_ABI unsigned getVGPREncodingGranule(Triple::SubArchType SubArch,
LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32);
LLVM_ABI unsigned getTotalNumVGPRs(Triple::SubArchType SubArch, bool IsWave32);
+/// Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
+constexpr unsigned MaxDynamicVGPRBlocks = 8;
+
/// \returns Number of VGPRs a single wave can address. On a target with a
-/// unified register file this covers the AGPRs as well. This does not account
-/// for dynamic VGPR mode, which caps allocation at a fixed number of blocks.
-LLVM_ABI unsigned getAddressableNumVGPRs(GPUKind AK, bool IsWave32);
+/// unified register file this covers the AGPRs as well. A nonzero
+/// \p DynamicVGPRBlockSize selects dynamic VGPR mode, which caps allocation at
+/// \c MaxDynamicVGPRBlocks blocks. On gfx90a-family targets, the unified
+/// register file size is returned regardless of \p DynamicVGPRBlockSize.
+LLVM_ABI unsigned getAddressableNumVGPRs(GPUKind AK, bool IsWave32,
+ unsigned DynamicVGPRBlockSize = 0);
LLVM_ABI unsigned getAddressableNumVGPRs(Triple::SubArchType SubArch,
- bool IsWave32);
+ bool IsWave32,
+ unsigned DynamicVGPRBlockSize = 0);
/// LDS size queries.
///
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index cbe54f4d89832..d038f1a442f64 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1161,13 +1161,12 @@ void AMDGPUAsmPrinter::emitDVgprSymbol(MachineFunction &MF) {
unsigned NumBlocks =
divideCeil(std::max(unsigned(NumVGPRs.getConstant()), 1U), BlockSize);
- if (NumBlocks > AMDGPU::IsaInfo::MaxDynamicVGPRBlocks) {
- OutContext.reportError(
- {}, "DVGPR block count " + Twine(NumBlocks) +
- " exceeds maximum of " +
- Twine(AMDGPU::IsaInfo::MaxDynamicVGPRBlocks) +
- " for __dvgpr$ symbol for '" +
- Twine(CurrentFnSym->getName()) + "'");
+ if (NumBlocks > AMDGPU::MaxDynamicVGPRBlocks) {
+ OutContext.reportError({}, "DVGPR block count " + Twine(NumBlocks) +
+ " exceeds maximum of " +
+ Twine(AMDGPU::MaxDynamicVGPRBlocks) +
+ " for __dvgpr$ symbol for '" +
+ Twine(CurrentFnSym->getName()) + "'");
return;
}
unsigned EncodedNumBlocks = (NumBlocks - 1) << 3;
diff --git a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
index 460c872dc46d6..fd7d643404e39 100644
--- a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
@@ -170,8 +170,7 @@ void GCNSchedStrategy::initialize(ScheduleDAGMI *DAG) {
"VGPRCriticalLimit calculation method.\n");
unsigned DynamicVGPRBlockSize = MFI.getDynamicVGPRBlockSize();
unsigned Granule = ST.getVGPRAllocGranule(DynamicVGPRBlockSize);
- unsigned Addressable =
- AMDGPU::IsaInfo::getAddressableNumVGPRs(ST, DynamicVGPRBlockSize);
+ unsigned Addressable = ST.getAddressableNumVGPRs(DynamicVGPRBlockSize);
unsigned VGPRBudget = alignDown(Addressable / TargetOccupancy, Granule);
VGPRBudget = std::max(VGPRBudget, Granule);
VGPRCriticalLimit = std::min(VGPRBudget, VGPRExcessLimit);
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 437cc9e942d5a..d5660a2c9fb04 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -871,14 +871,8 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
/// \returns Addressable number of VGPRs supported by the subtarget.
unsigned getAddressableNumVGPRs(unsigned DynamicVGPRBlockSize) const {
- // Dynamic VGPR mode is a per-kernel mode, so it is not covered by the
- // TargetParser query.
- if (DynamicVGPRBlockSize != 0) {
- return AMDGPU::IsaInfo::getAddressableNumVGPRs(*this,
- DynamicVGPRBlockSize);
- }
return AMDGPU::getAddressableNumVGPRs(getTargetID().getGPUKind(),
- isWave32());
+ isWave32(), DynamicVGPRBlockSize);
}
/// \returns the minimum number of VGPRs that will prevent achieving more than
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 82a9603915291..63ece72e08879 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1307,19 +1307,6 @@ unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI) {
return 256;
}
-unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
- unsigned DynamicVGPRBlockSize) {
- const auto &Features = STI.getFeatureBits();
- if (Features.test(FeatureGFX90AInsts))
- return 512;
-
- if (DynamicVGPRBlockSize != 0) {
- // On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
- return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
- }
- return getAddressableNumArchVGPRs(STI);
-}
-
unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI,
unsigned NumVGPRs,
unsigned DynamicVGPRBlockSize) {
@@ -1382,7 +1369,7 @@ unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
unsigned TotNumVGPRs = AMDGPU::getTotalNumVGPRs(Kind, IsWave32);
unsigned AddrsableNumVGPRs =
- getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
+ AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, DynamicVGPRBlockSize);
unsigned Granule =
AMDGPU::getVGPRAllocGranule(Kind, IsWave32, DynamicVGPRBlockSize);
unsigned MaxNumVGPRs = alignDown(TotNumVGPRs / WavesPerEU, Granule);
@@ -1416,7 +1403,7 @@ unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
AMDGPU::getVGPRAllocGranule(
Kind, IsWave32, DynamicVGPRBlockSize));
unsigned AddressableNumVGPRs =
- getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
+ AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, DynamicVGPRBlockSize);
return std::min(MaxNumVGPRs, AddressableNumVGPRs);
}
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index a57cfee284738..211771a8dab11 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -237,17 +237,10 @@ unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs);
/// returns the allocation granule for ArchVGPRs.
unsigned getArchVGPRAllocGranule();
-/// Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
-static constexpr unsigned MaxDynamicVGPRBlocks = 8;
-
/// \returns Addressable number of architectural VGPRs for a given subtarget \p
/// STI.
unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI);
-/// \returns Addressable number of VGPRs for given subtarget \p STI.
-unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
- unsigned DynamicVGPRBlockSize);
-
/// \returns Minimum number of VGPRs that meets given number of waves per
/// execution unit requirement for given subtarget \p STI.
unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index bcdf5c356ef98..d82a82bd9699e 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -471,19 +471,24 @@ unsigned AMDGPU::getTotalNumVGPRs(Triple::SubArchType SubArch, bool IsWave32) {
return getTotalNumVGPRs(getGPUKindFromSubArch(SubArch), IsWave32);
}
-unsigned AMDGPU::getAddressableNumVGPRs(GPUKind AK, bool IsWave32) {
+unsigned AMDGPU::getAddressableNumVGPRs(GPUKind AK, bool IsWave32,
+ unsigned DynamicVGPRBlockSize) {
const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
// The unified register file makes the AGPRs addressable as VGPRs.
if (Features.test(FEAT_GFX90A_INSTS))
return 512;
+ if (DynamicVGPRBlockSize != 0)
+ return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
if (Features.test(FEAT_1024_ADDRESSABLE_VGPRS))
return IsWave32 ? 1024 : 512;
return 256;
}
unsigned AMDGPU::getAddressableNumVGPRs(Triple::SubArchType SubArch,
- bool IsWave32) {
- return getAddressableNumVGPRs(getGPUKindFromSubArch(SubArch), IsWave32);
+ bool IsWave32,
+ unsigned DynamicVGPRBlockSize) {
+ return getAddressableNumVGPRs(getGPUKindFromSubArch(SubArch), IsWave32,
+ DynamicVGPRBlockSize);
}
unsigned AMDGPU::getMaxHWAddressableLocalMemorySize(GPUKind AK) {
diff --git a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
index a21165d041bc6..2ef08e4bb70d1 100644
--- a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
+++ b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
@@ -218,8 +218,7 @@ int main(int argc, char **argv) {
unsigned NumWorkGroupSIMDs = ST.getNumWorkGroupSIMDs();
unsigned LocalMemSize = AMDGPU::IsaInfo::getLocalMemorySize(STI);
unsigned AddrLocalMem = AMDGPU::IsaInfo::getAddressableLocalMemorySize(STI);
- unsigned AddrVGPRs =
- AMDGPU::IsaInfo::getAddressableNumVGPRs(STI, DynVGPRBlockSize);
+ unsigned AddrVGPRs = ST.getAddressableNumVGPRs(DynVGPRBlockSize);
unsigned AddrSGPRs = ST.getAddressableNumSGPRs();
unsigned MaxWGSize = AMDGPU::getMaxFlatWorkGroupSize();
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 7c5f156775bd2..1844f929fd849 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -3327,6 +3327,37 @@ TEST(TargetParserTest, testAMDGPUDynamicVGPRAllocGranule) {
8u);
}
+TEST(TargetParserTest, testAMDGPUDynamicVGPRAddressableNum) {
+ for (auto Kind :
+ {AMDGPU::GK_GFX1200, AMDGPU::GK_GFX1250, AMDGPU::GK_GFX1310}) {
+ SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+ for (bool IsWave32 : {false, true}) {
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, 16), 128u);
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, 32), 256u);
+ }
+ }
+
+ // Zero selects the static limits, which depend on the wavefront size.
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(AMDGPU::GK_GFX1250, false, 0), 512u);
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(AMDGPU::GK_GFX1250, true, 0), 1024u);
+
+ // gfx90a-family targets always use their unified register file size.
+ for (auto Kind : {AMDGPU::GK_GFX90A, AMDGPU::GK_GFX942, AMDGPU::GK_GFX950}) {
+ SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Kind).str());
+ for (bool IsWave32 : {false, true}) {
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Kind, IsWave32, 16), 512u);
+ }
+ }
+
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Triple::AMDGPUSubArch1200, true, 16),
+ 128u);
+ EXPECT_EQ(
+ AMDGPU::getAddressableNumVGPRs(Triple::AMDGPUSubArch1250, false, 32),
+ 256u);
+ EXPECT_EQ(AMDGPU::getAddressableNumVGPRs(Triple::AMDGPUSubArch90A, false, 16),
+ 512u);
+}
+
TEST(TargetParserTest, testAMDGPUgetMaxHWAddressableLocalMemorySize) {
// The addressable cap is a fixed hardware property, independent of how many
// SIMDs a work-group runs on.
>From cb3f6a2d8eec070354bbb82e77f67207d3e7ba9f Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sun, 27 Sep 2026 14:17:08 -0400
Subject: [PATCH 4/4] [AMDGPU] Remove remaining BaseInfo duplicates of
TargetParser APIs
Use TargetParser for LDS sizes and architectural VGPR limits, preserving
the distinction between VGPRs and AGPRs on unified-register-file targets.
Remove the TargetID factory wrapper and redundant type and SGPR constant
aliases, updating the remaining callers.
Cover physical LDS allocation boundaries across GPU generations and
full/half-SIMD modes in the assembler tests.
Change-Id: I9fa8447a69de3649792809424d2b0251dae0155b
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 5 +-
.../AMDGPU/AsmParser/AMDGPUAsmParser.cpp | 6 +-
.../Disassembler/AMDGPUDisassembler.cpp | 3 +-
llvm/lib/Target/AMDGPU/GCNSubtarget.cpp | 4 +-
llvm/lib/Target/AMDGPU/GCNSubtarget.h | 7 +-
.../MCTargetDesc/AMDGPUTargetStreamer.cpp | 5 +-
.../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 67 -------------------
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 24 +------
llvm/test/MC/AMDGPU/elf-lds-size.s | 15 +++++
.../llvm-calc-occupancy.cpp | 4 +-
10 files changed, 37 insertions(+), 103 deletions(-)
create mode 100644 llvm/test/MC/AMDGPU/elf-lds-size.s
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index d038f1a442f64..eb105c9d22afa 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1389,10 +1389,9 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
}
if (STM.hasSGPRInitBug()) {
- ProgInfo.NumSGPR =
- CreateExpr(AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG);
+ ProgInfo.NumSGPR = CreateExpr(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG);
ProgInfo.NumSGPRsForWavesPerEU =
- CreateExpr(AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG);
+ CreateExpr(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG);
}
if (MFI->getNumUserSGPRs() > STM.getMaxNumUserSGPRs()) {
diff --git a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
index 3a7213e52d5f5..62002de8fbc3f 100644
--- a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
+++ b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
@@ -6175,7 +6175,7 @@ bool AMDGPUAsmParser::calculateGPRBlocks(
if (Features.test(FeatureSGPRInitBug))
NumSGPRs =
- MCConstantExpr::create(IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG, Ctx);
+ MCConstantExpr::create(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG, Ctx);
}
// The MCExpr equivalent of getNumSGPRBlocks/getNumVGPRBlocks:
@@ -6998,7 +6998,9 @@ bool AMDGPUAsmParser::ParseDirectiveAMDGPULDS() {
if (getParser().parseComma())
return true;
- unsigned LocalMemorySize = AMDGPU::IsaInfo::getLocalMemorySize(getSTI());
+ unsigned LocalMemorySize =
+ AMDGPU::getLocalMemorySize(AMDGPU::parseArchAMDGCN(getSTI().getCPU()),
+ AMDGPU::isFullSIMDMode(getSTI()));
int64_t Size;
SMLoc SizeLoc = getLoc();
diff --git a/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp b/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
index 8d930b3761859..73f0bae68d81d 100644
--- a/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
+++ b/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
@@ -60,7 +60,8 @@ AMDGPUDisassembler::AMDGPUDisassembler(const MCSubtargetInfo &STI,
MAI(Ctx.getAsmInfo()),
HwModeRegClass(STI.getHwMode(MCSubtargetInfo::HwMode_RegInfo)),
TargetMaxInstBytes(MAI.getMaxInstLength(&STI)),
- TargetID(AMDGPU::createAMDGPUTargetID(STI, "")),
+ TargetID(AMDGPU::TargetID::createFromSubtargetFeatures(
+ STI.getTargetTriple(), STI.getCPU(), "")),
CodeObjectVersion(AMDGPU::getDefaultAMDHSACodeObjectVersion()) {
// ToDo: AMDGPUDisassembler supports only VI ISA.
if (!STI.hasFeature(AMDGPU::FeatureGCN3Encoding) && !isGFX10Plus())
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
index 34d987cd6d266..5c0bc79e8dbd4 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
@@ -223,7 +223,7 @@ GCNSubtarget::GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS,
: // clang-format off
AMDGPUGenSubtargetInfo(TT, GPU, /*TuneCPU*/ GPU, FS),
AMDGPUSubtarget(TT),
- TargetID(AMDGPU::createAMDGPUTargetID(*this, "")),
+ TargetID(AMDGPU::TargetID::createFromSubtargetFeatures(TT, GPU, "")),
InstrItins(getInstrItineraryForCPU(GPU)),
BufferOOBRelaxed(BufferOOBRelaxed),
TBufferOOBRelaxed(TBufferOOBRelaxed),
@@ -576,7 +576,7 @@ unsigned GCNSubtarget::getBaseMaxNumSGPRs(
}
if (hasSGPRInitBug())
- MaxNumSGPRs = AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG;
+ MaxNumSGPRs = AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG;
return std::min(MaxNumSGPRs - ReservedNumSGPRs, MaxAddressableNumSGPRs);
}
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index d5660a2c9fb04..0dd681dceeb2a 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -866,7 +866,12 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
/// \returns Addressable number of architectural VGPRs supported by the
/// subtarget.
unsigned getAddressableNumArchVGPRs() const {
- return AMDGPU::IsaInfo::getAddressableNumArchVGPRs(*this);
+ // The TargetParser query includes AGPRs on targets with a unified register
+ // file, but only 256 registers are architectural VGPRs.
+ if (hasGFX90AInsts())
+ return 256;
+ return AMDGPU::getAddressableNumVGPRs(getTargetID().getGPUKind(),
+ isWave32());
}
/// \returns Addressable number of VGPRs supported by the subtarget.
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
index 6b1e58b52f305..68490fc64f6e0 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
@@ -51,8 +51,9 @@ void AMDGPUTargetStreamer::initializeTargetID(const MCSubtargetInfo &STI,
assert(TargetID == std::nullopt && "TargetID can only be initialized once");
// Apply xnack/sramecc from subtarget features only in MC contexts
// (assembler), not in codegen where they come from module flags
- TargetID = AMDGPU::createAMDGPUTargetID(
- STI, ApplyFeatureString ? STI.getFeatureString() : "");
+ TargetID = AMDGPU::TargetID::createFromSubtargetFeatures(
+ STI.getTargetTriple(), STI.getCPU(),
+ ApplyFeatureString ? STI.getFeatureString() : "");
}
bool AMDGPUTargetStreamer::EmitHSAMetadataV3(StringRef HSAMetadataString) {
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 63ece72e08879..40577166f8d27 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1088,15 +1088,6 @@ VOPD::InstInfo getVOPDInstInfo(unsigned VOPDOpcode,
return VOPD::InstInfo(OpXInfo, OpYInfo);
}
-TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI,
- StringRef FeatureString) {
- // In codegen the mode comes from module flags and FeatureString is empty, so
- // the processor defaults apply. The assembler has no target directive, so it
- // pins the mode via the +xnack/-xnack/+sramecc/-sramecc feature string.
- return TargetID::createFromSubtargetFeatures(STI.getTargetTriple(),
- STI.getCPU(), FeatureString);
-}
-
namespace IsaInfo {
unsigned getInstCacheLineSize(const MCSubtargetInfo &STI) {
@@ -1116,57 +1107,6 @@ unsigned getWavefrontSize(const MCSubtargetInfo &STI) {
return 64;
}
-// Maximum LDS a single work-group can address. This is a fixed HW cap. It does
-// not depend on how many SIMDs a work-group runs on.
-static unsigned getMaxHWAddressableLocalMemorySize(const MCSubtargetInfo &STI) {
- if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
- return 32768;
- if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
- return 65536;
- if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
- return 163840;
- if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
- return 196608;
- if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
- return 327680;
- return 32768;
-}
-
-// Total physical size of LDS on the block, in bytes. On targets with
-// FeatureHalfAddressablePhysicalLocalMemory the physical block is twice the
-// addressable size (gfx6: 64 KiB physical and 32 KiB addressable;
-// gfx10/11/12: 128 KiB physical and 64 KiB addressable). On other targets it is
-// equal to the addressable size.
-static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI) {
- unsigned Addressable = getMaxHWAddressableLocalMemorySize(STI);
- if (STI.getFeatureBits().test(FeatureHalfAddressablePhysicalLocalMemory))
- return 2 * Addressable;
- return Addressable;
-}
-
-// Sizes in use, by generation (addressable / physical block):
-// gfx6 : 32 KiB addressable, 64 KiB physical block
-// gfx7 / gfx8 / gfx9: 64 KiB
-// gfx9.5 (gfx950) : 160 KiB
-// gfx10 / 11 / 12 : 64 KiB addressable, 128 KiB physical block
-// gfx12.5 (gfx1250) : 320 KiB (always runs on four SIMDs)
-// gfx13 : 192 KiB on four SIMDs, 96 KiB on two
-// Total available in the current mode. The physical size is halved when a
-// work-group runs on two SIMDs.
-unsigned getLocalMemorySize(const MCSubtargetInfo &STI) {
- unsigned Size = getPhysicalLocalMemorySize(STI);
- if (!isFullSIMDMode(STI))
- Size /= 2;
- return Size;
-}
-
-// What one work-group can allocate in the current mode. This is the HW
-// addressable cap, but never more than the total available in the current mode.
-unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI) {
- return std::min(getMaxHWAddressableLocalMemorySize(STI),
- getLocalMemorySize(STI));
-}
-
unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI,
unsigned FlatWorkGroupSize) {
assert(FlatWorkGroupSize != 0);
@@ -1300,13 +1240,6 @@ unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs) {
unsigned getArchVGPRAllocGranule() { return 4; }
-unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI) {
- const auto &Features = STI.getFeatureBits();
- if (Features.test(Feature1024AddressableVGPRs))
- return Features.test(FeatureWavefrontSize32) ? 1024 : 512;
- return 256;
-}
-
unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI,
unsigned NumVGPRs,
unsigned DynamicVGPRBlockSize) {
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index 211771a8dab11..50f3185ed234b 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -161,20 +161,9 @@ struct WMMAInstInfo {
#define GET_WMMAInstInfoTable_DECL
#include "AMDGPUGenSearchableTables.inc"
-using TargetIDSetting = AMDGPU::TargetIDSetting;
-using TargetID = AMDGPU::TargetID;
-
-/// Construct TargetID from MCSubtargetInfo. \p FeatureString is used to
-/// determine explicitly requested xnack/sramecc settings.
-TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI,
- StringRef FeatureString);
-
namespace IsaInfo {
-enum {
- FIXED_NUM_SGPRS_FOR_INIT_BUG = AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG,
- TRAP_NUM_SGPRS = 16
-};
+enum { TRAP_NUM_SGPRS = 16 };
/// Returns true if \p Lhs and \p Rhs are incompatible (both specific but
/// different).
@@ -189,13 +178,6 @@ unsigned getInstCacheLineSize(const MCSubtargetInfo &STI);
/// \returns Wavefront size for given subtarget \p STI.
unsigned getWavefrontSize(const MCSubtargetInfo &STI);
-/// \returns Local memory size in bytes for given subtarget \p STI.
-unsigned getLocalMemorySize(const MCSubtargetInfo &STI);
-
-/// \returns Maximum addressable local memory size in bytes for given subtarget
-/// \p STI.
-unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI);
-
/// \returns Maximum number of work groups per compute unit for given subtarget
/// \p STI and limited by given \p FlatWorkGroupSize.
unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI,
@@ -237,10 +219,6 @@ unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs);
/// returns the allocation granule for ArchVGPRs.
unsigned getArchVGPRAllocGranule();
-/// \returns Addressable number of architectural VGPRs for a given subtarget \p
-/// STI.
-unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI);
-
/// \returns Minimum number of VGPRs that meets given number of waves per
/// execution unit requirement for given subtarget \p STI.
unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
diff --git a/llvm/test/MC/AMDGPU/elf-lds-size.s b/llvm/test/MC/AMDGPU/elf-lds-size.s
new file mode 100644
index 0000000000000..88b45e75316e6
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/elf-lds-size.s
@@ -0,0 +1,15 @@
+// RUN: not llvm-mc -triple=amdgpu6.00-- -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu9.00-- -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu9.50-- -defsym LDS_SIZE=163840 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu10.30-- -mattr=-cumode -defsym LDS_SIZE=131072 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu10.30-- -mattr=+cumode -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu12.50-- -mattr=-cumode -defsym LDS_SIZE=327680 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu12.50-- -mattr=+cumode -defsym LDS_SIZE=327680 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu13.10-- -mattr=-cumode -defsym LDS_SIZE=196608 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu13.10-- -mattr=+cumode -defsym LDS_SIZE=98304 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+
+// LDS symbols can use the physical LDS available in the current mode, even
+// when it is larger than the amount a single work-group can address.
+.amdgpu_lds at_limit, LDS_SIZE
+.amdgpu_lds over_limit, LDS_SIZE + 1
+// CHECK: :[[@LINE-1]]:25: error: size is too large
diff --git a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
index 2ef08e4bb70d1..55786600554a9 100644
--- a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
+++ b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
@@ -216,8 +216,8 @@ int main(int argc, char **argv) {
unsigned WaveSize = AMDGPU::IsaInfo::getWavefrontSize(STI);
unsigned MaxWaves = ST.getMaxWavesPerEU();
unsigned NumWorkGroupSIMDs = ST.getNumWorkGroupSIMDs();
- unsigned LocalMemSize = AMDGPU::IsaInfo::getLocalMemorySize(STI);
- unsigned AddrLocalMem = AMDGPU::IsaInfo::getAddressableLocalMemorySize(STI);
+ unsigned LocalMemSize = ST.getLocalMemorySize();
+ unsigned AddrLocalMem = ST.getAddressableLocalMemorySize();
unsigned AddrVGPRs = ST.getAddressableNumVGPRs(DynVGPRBlockSize);
unsigned AddrSGPRs = ST.getAddressableNumSGPRs();
unsigned MaxWGSize = AMDGPU::getMaxFlatWorkGroupSize();
More information about the llvm-commits
mailing list