[llvm] [AMDGPU] Remove BaseInfo duplicates of TargetParser APIs (PR #226744)
via llvm-commits
llvm-commits at lists.llvm.org
Sat Sep 26 19:27:29 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Chinmay Deshpande (chinmaydd)
<details>
<summary>Changes</summary>
Use TargetParser for LDS sizes and VGPR hardware limits, retaining the backend handling of dynamic VGPR allocation and architectural VGPRs on targets with a unified register file.
---
Full diff: https://github.com/llvm/llvm-project/pull/226744.diff
10 Files Affected:
- (modified) llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp (+2-3)
- (modified) llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp (+4-2)
- (modified) llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp (+2-1)
- (modified) llvm/lib/Target/AMDGPU/GCNSubtarget.cpp (+2-2)
- (modified) llvm/lib/Target/AMDGPU/GCNSubtarget.h (+6-1)
- (modified) llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp (+3-2)
- (modified) llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp (+9-86)
- (modified) llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h (+1-23)
- (added) llvm/test/MC/AMDGPU/elf-lds-size.s (+15)
- (modified) llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp (+2-2)
``````````diff
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index c18b840c36791..7650d8b5bf15f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1390,10 +1390,9 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
}
if (STM.hasSGPRInitBug()) {
- ProgInfo.NumSGPR =
- CreateExpr(AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG);
+ ProgInfo.NumSGPR = CreateExpr(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG);
ProgInfo.NumSGPRsForWavesPerEU =
- CreateExpr(AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG);
+ CreateExpr(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG);
}
if (MFI->getNumUserSGPRs() > STM.getMaxNumUserSGPRs()) {
diff --git a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
index 812acc8f8be85..88d5f5de25312 100644
--- a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
+++ b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
@@ -6175,7 +6175,7 @@ bool AMDGPUAsmParser::calculateGPRBlocks(
if (Features.test(FeatureSGPRInitBug))
NumSGPRs =
- MCConstantExpr::create(IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG, Ctx);
+ MCConstantExpr::create(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG, Ctx);
}
// The MCExpr equivalent of getNumSGPRBlocks/getNumVGPRBlocks:
@@ -6997,7 +6997,9 @@ bool AMDGPUAsmParser::ParseDirectiveAMDGPULDS() {
if (getParser().parseComma())
return true;
- unsigned LocalMemorySize = AMDGPU::IsaInfo::getLocalMemorySize(getSTI());
+ unsigned LocalMemorySize =
+ AMDGPU::getLocalMemorySize(AMDGPU::parseArchAMDGCN(getSTI().getCPU()),
+ AMDGPU::isFullSIMDMode(getSTI()));
int64_t Size;
SMLoc SizeLoc = getLoc();
diff --git a/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp b/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
index 3b200d412eb0d..1a5223e3cffab 100644
--- a/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
+++ b/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
@@ -60,7 +60,8 @@ AMDGPUDisassembler::AMDGPUDisassembler(const MCSubtargetInfo &STI,
MAI(Ctx.getAsmInfo()),
HwModeRegClass(STI.getHwMode(MCSubtargetInfo::HwMode_RegInfo)),
TargetMaxInstBytes(MAI.getMaxInstLength(&STI)),
- TargetID(AMDGPU::createAMDGPUTargetID(STI, "")),
+ TargetID(AMDGPU::TargetID::createFromSubtargetFeatures(
+ STI.getTargetTriple(), STI.getCPU(), "")),
CodeObjectVersion(AMDGPU::getDefaultAMDHSACodeObjectVersion()) {
// ToDo: AMDGPUDisassembler supports only VI ISA.
if (!STI.hasFeature(AMDGPU::FeatureGCN3Encoding) && !isGFX10Plus())
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
index 34d987cd6d266..5c0bc79e8dbd4 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
@@ -223,7 +223,7 @@ GCNSubtarget::GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS,
: // clang-format off
AMDGPUGenSubtargetInfo(TT, GPU, /*TuneCPU*/ GPU, FS),
AMDGPUSubtarget(TT),
- TargetID(AMDGPU::createAMDGPUTargetID(*this, "")),
+ TargetID(AMDGPU::TargetID::createFromSubtargetFeatures(TT, GPU, "")),
InstrItins(getInstrItineraryForCPU(GPU)),
BufferOOBRelaxed(BufferOOBRelaxed),
TBufferOOBRelaxed(TBufferOOBRelaxed),
@@ -576,7 +576,7 @@ unsigned GCNSubtarget::getBaseMaxNumSGPRs(
}
if (hasSGPRInitBug())
- MaxNumSGPRs = AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG;
+ MaxNumSGPRs = AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG;
return std::min(MaxNumSGPRs - ReservedNumSGPRs, MaxAddressableNumSGPRs);
}
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 2e09f1f1fda39..c8020be9475c6 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -864,7 +864,12 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
/// \returns Addressable number of architectural VGPRs supported by the
/// subtarget.
unsigned getAddressableNumArchVGPRs() const {
- return AMDGPU::IsaInfo::getAddressableNumArchVGPRs(*this);
+ // The TargetParser query includes AGPRs on targets with a unified register
+ // file, but only 256 registers are architectural VGPRs.
+ if (hasGFX90AInsts())
+ return 256;
+ return AMDGPU::getAddressableNumVGPRs(getTargetID().getGPUKind(),
+ isWave32());
}
/// \returns Addressable number of VGPRs supported by the subtarget.
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
index 6b1e58b52f305..68490fc64f6e0 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
@@ -51,8 +51,9 @@ void AMDGPUTargetStreamer::initializeTargetID(const MCSubtargetInfo &STI,
assert(TargetID == std::nullopt && "TargetID can only be initialized once");
// Apply xnack/sramecc from subtarget features only in MC contexts
// (assembler), not in codegen where they come from module flags
- TargetID = AMDGPU::createAMDGPUTargetID(
- STI, ApplyFeatureString ? STI.getFeatureString() : "");
+ TargetID = AMDGPU::TargetID::createFromSubtargetFeatures(
+ STI.getTargetTriple(), STI.getCPU(),
+ ApplyFeatureString ? STI.getFeatureString() : "");
}
bool AMDGPUTargetStreamer::EmitHSAMetadataV3(StringRef HSAMetadataString) {
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index e45359719d6c9..c56d6599d54ac 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1088,15 +1088,6 @@ VOPD::InstInfo getVOPDInstInfo(unsigned VOPDOpcode,
return VOPD::InstInfo(OpXInfo, OpYInfo);
}
-TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI,
- StringRef FeatureString) {
- // In codegen the mode comes from module flags and FeatureString is empty, so
- // the processor defaults apply. The assembler has no target directive, so it
- // pins the mode via the +xnack/-xnack/+sramecc/-sramecc feature string.
- return TargetID::createFromSubtargetFeatures(STI.getTargetTriple(),
- STI.getCPU(), FeatureString);
-}
-
namespace IsaInfo {
unsigned getInstCacheLineSize(const MCSubtargetInfo &STI) {
@@ -1116,57 +1107,6 @@ unsigned getWavefrontSize(const MCSubtargetInfo &STI) {
return 64;
}
-// Maximum LDS a single work-group can address. This is a fixed HW cap. It does
-// not depend on how many SIMDs a work-group runs on.
-static unsigned getMaxHWAddressableLocalMemorySize(const MCSubtargetInfo &STI) {
- if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
- return 32768;
- if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
- return 65536;
- if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
- return 163840;
- if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
- return 196608;
- if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
- return 327680;
- return 32768;
-}
-
-// Total physical size of LDS on the block, in bytes. On targets with
-// FeatureHalfAddressablePhysicalLocalMemory the physical block is twice the
-// addressable size (gfx6: 64 KiB physical and 32 KiB addressable;
-// gfx10/11/12: 128 KiB physical and 64 KiB addressable). On other targets it is
-// equal to the addressable size.
-static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI) {
- unsigned Addressable = getMaxHWAddressableLocalMemorySize(STI);
- if (STI.getFeatureBits().test(FeatureHalfAddressablePhysicalLocalMemory))
- return 2 * Addressable;
- return Addressable;
-}
-
-// Sizes in use, by generation (addressable / physical block):
-// gfx6 : 32 KiB addressable, 64 KiB physical block
-// gfx7 / gfx8 / gfx9: 64 KiB
-// gfx9.5 (gfx950) : 160 KiB
-// gfx10 / 11 / 12 : 64 KiB addressable, 128 KiB physical block
-// gfx12.5 (gfx1250) : 320 KiB (always runs on four SIMDs)
-// gfx13 : 192 KiB on four SIMDs, 96 KiB on two
-// Total available in the current mode. The physical size is halved when a
-// work-group runs on two SIMDs.
-unsigned getLocalMemorySize(const MCSubtargetInfo &STI) {
- unsigned Size = getPhysicalLocalMemorySize(STI);
- if (!isFullSIMDMode(STI))
- Size /= 2;
- return Size;
-}
-
-// What one work-group can allocate in the current mode. This is the HW
-// addressable cap, but never more than the total available in the current mode.
-unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI) {
- return std::min(getMaxHWAddressableLocalMemorySize(STI),
- getLocalMemorySize(STI));
-}
-
unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI,
unsigned FlatWorkGroupSize) {
assert(FlatWorkGroupSize != 0);
@@ -1301,23 +1241,15 @@ unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs) {
unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI,
unsigned DynamicVGPRBlockSize,
std::optional<bool> EnableWavefrontSize32) {
- if (STI.getFeatureBits().test(FeatureGFX90AInsts))
- return 8;
-
- if (DynamicVGPRBlockSize != 0)
+ if (DynamicVGPRBlockSize != 0 &&
+ !STI.getFeatureBits().test(FeatureGFX90AInsts))
return DynamicVGPRBlockSize;
bool IsWave32 = EnableWavefrontSize32
? *EnableWavefrontSize32
: STI.getFeatureBits().test(FeatureWavefrontSize32);
- if (STI.getFeatureBits().test(Feature1536VGPRs))
- return IsWave32 ? 24 : 12;
-
- if (hasGFX10_3Insts(STI))
- return IsWave32 ? 16 : 8;
-
- return IsWave32 ? 8 : 4;
+ return AMDGPU::getVGPRAllocGranule(parseArchAMDGCN(STI.getCPU()), IsWave32);
}
unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI,
@@ -1337,25 +1269,16 @@ unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI,
unsigned getArchVGPRAllocGranule() { return 4; }
-unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI) {
- const auto &Features = STI.getFeatureBits();
- if (Features.test(Feature1024AddressableVGPRs))
- return Features.test(FeatureWavefrontSize32) ? 1024 : 512;
- return 256;
-}
-
unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
unsigned DynamicVGPRBlockSize) {
- const auto &Features = STI.getFeatureBits();
- if (Features.test(FeatureGFX90AInsts))
- return 512;
-
- if (DynamicVGPRBlockSize != 0) {
+ if (DynamicVGPRBlockSize != 0 &&
+ !STI.getFeatureBits().test(FeatureGFX90AInsts)) {
// On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
- return MaxDynamicVGPRBlocks *
- getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
+ return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
}
- return getAddressableNumArchVGPRs(STI);
+ return AMDGPU::getAddressableNumVGPRs(
+ parseArchAMDGCN(STI.getCPU()),
+ STI.getFeatureBits().test(FeatureWavefrontSize32));
}
unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI,
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index bce059a0c18a7..b4a37c6fe4be4 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -161,20 +161,9 @@ struct WMMAInstInfo {
#define GET_WMMAInstInfoTable_DECL
#include "AMDGPUGenSearchableTables.inc"
-using TargetIDSetting = AMDGPU::TargetIDSetting;
-using TargetID = AMDGPU::TargetID;
-
-/// Construct TargetID from MCSubtargetInfo. \p FeatureString is used to
-/// determine explicitly requested xnack/sramecc settings.
-TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI,
- StringRef FeatureString);
-
namespace IsaInfo {
-enum {
- FIXED_NUM_SGPRS_FOR_INIT_BUG = AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG,
- TRAP_NUM_SGPRS = 16
-};
+enum { TRAP_NUM_SGPRS = 16 };
/// Returns true if \p Lhs and \p Rhs are incompatible (both specific but
/// different).
@@ -189,13 +178,6 @@ unsigned getInstCacheLineSize(const MCSubtargetInfo &STI);
/// \returns Wavefront size for given subtarget \p STI.
unsigned getWavefrontSize(const MCSubtargetInfo &STI);
-/// \returns Local memory size in bytes for given subtarget \p STI.
-unsigned getLocalMemorySize(const MCSubtargetInfo &STI);
-
-/// \returns Maximum addressable local memory size in bytes for given subtarget
-/// \p STI.
-unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI);
-
/// \returns Maximum number of work groups per compute unit for given subtarget
/// \p STI and limited by given \p FlatWorkGroupSize.
unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI,
@@ -256,10 +238,6 @@ unsigned getArchVGPRAllocGranule();
/// Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
static constexpr unsigned MaxDynamicVGPRBlocks = 8;
-/// \returns Addressable number of architectural VGPRs for a given subtarget \p
-/// STI.
-unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI);
-
/// \returns Addressable number of VGPRs for given subtarget \p STI.
unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
unsigned DynamicVGPRBlockSize);
diff --git a/llvm/test/MC/AMDGPU/elf-lds-size.s b/llvm/test/MC/AMDGPU/elf-lds-size.s
new file mode 100644
index 0000000000000..88b45e75316e6
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/elf-lds-size.s
@@ -0,0 +1,15 @@
+// RUN: not llvm-mc -triple=amdgpu6.00-- -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu9.00-- -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu9.50-- -defsym LDS_SIZE=163840 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu10.30-- -mattr=-cumode -defsym LDS_SIZE=131072 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu10.30-- -mattr=+cumode -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu12.50-- -mattr=-cumode -defsym LDS_SIZE=327680 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu12.50-- -mattr=+cumode -defsym LDS_SIZE=327680 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu13.10-- -mattr=-cumode -defsym LDS_SIZE=196608 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu13.10-- -mattr=+cumode -defsym LDS_SIZE=98304 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+
+// LDS symbols can use the physical LDS available in the current mode, even
+// when it is larger than the amount a single work-group can address.
+.amdgpu_lds at_limit, LDS_SIZE
+.amdgpu_lds over_limit, LDS_SIZE + 1
+// CHECK: :[[@LINE-1]]:25: error: size is too large
diff --git a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
index a21165d041bc6..2994fb2c3ceb2 100644
--- a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
+++ b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
@@ -216,8 +216,8 @@ int main(int argc, char **argv) {
unsigned WaveSize = AMDGPU::IsaInfo::getWavefrontSize(STI);
unsigned MaxWaves = ST.getMaxWavesPerEU();
unsigned NumWorkGroupSIMDs = ST.getNumWorkGroupSIMDs();
- unsigned LocalMemSize = AMDGPU::IsaInfo::getLocalMemorySize(STI);
- unsigned AddrLocalMem = AMDGPU::IsaInfo::getAddressableLocalMemorySize(STI);
+ unsigned LocalMemSize = ST.getLocalMemorySize();
+ unsigned AddrLocalMem = ST.getAddressableLocalMemorySize();
unsigned AddrVGPRs =
AMDGPU::IsaInfo::getAddressableNumVGPRs(STI, DynVGPRBlockSize);
unsigned AddrSGPRs = ST.getAddressableNumSGPRs();
``````````
</details>
https://github.com/llvm/llvm-project/pull/226744
More information about the llvm-commits
mailing list