[llvm] [AMDGPU] Remove BaseInfo duplicates of TargetParser APIs (PR #226744)

Chinmay Deshpande via llvm-commits llvm-commits at lists.llvm.org
Sat Sep 26 19:26:52 PDT 2026


https://github.com/chinmaydd created https://github.com/llvm/llvm-project/pull/226744

Use TargetParser for LDS sizes and VGPR hardware limits, retaining the backend handling of dynamic VGPR allocation and architectural VGPRs on targets with a unified register file.

>From 621339c4eda45c96d6d5008d9f3a0f0626934a3f Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sat, 26 Sep 2026 22:19:34 -0400
Subject: [PATCH] [AMDGPU] Remove BaseInfo duplicates of TargetParser APIs

Use TargetParser for LDS sizes and VGPR hardware limits, retaining the
backend handling of dynamic VGPR allocation and architectural VGPRs on
targets with a unified register file.

Remove the TargetID factory wrapper and redundant type and SGPR constant
aliases. Update the subtarget, assembler, disassembler, target streamer,
and occupancy calculator callers.

Cover physical LDS allocation boundaries across GPU generations and
full/half-SIMD modes in the assembler tests.

Change-Id: Iacae35d939d8bc1d5d78ad9e545c6f1eb5cd38fe
---
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp   |  5 +-
 .../AMDGPU/AsmParser/AMDGPUAsmParser.cpp      |  6 +-
 .../Disassembler/AMDGPUDisassembler.cpp       |  3 +-
 llvm/lib/Target/AMDGPU/GCNSubtarget.cpp       |  4 +-
 llvm/lib/Target/AMDGPU/GCNSubtarget.h         |  7 +-
 .../MCTargetDesc/AMDGPUTargetStreamer.cpp     |  5 +-
 .../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp    | 95 ++-----------------
 llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 24 +----
 llvm/test/MC/AMDGPU/elf-lds-size.s            | 15 +++
 .../llvm-calc-occupancy.cpp                   |  4 +-
 10 files changed, 46 insertions(+), 122 deletions(-)
 create mode 100644 llvm/test/MC/AMDGPU/elf-lds-size.s

diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index c18b840c36791..7650d8b5bf15f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1390,10 +1390,9 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
   }
 
   if (STM.hasSGPRInitBug()) {
-    ProgInfo.NumSGPR =
-        CreateExpr(AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG);
+    ProgInfo.NumSGPR = CreateExpr(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG);
     ProgInfo.NumSGPRsForWavesPerEU =
-        CreateExpr(AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG);
+        CreateExpr(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG);
   }
 
   if (MFI->getNumUserSGPRs() > STM.getMaxNumUserSGPRs()) {
diff --git a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
index 812acc8f8be85..88d5f5de25312 100644
--- a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
+++ b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
@@ -6175,7 +6175,7 @@ bool AMDGPUAsmParser::calculateGPRBlocks(
 
     if (Features.test(FeatureSGPRInitBug))
       NumSGPRs =
-          MCConstantExpr::create(IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG, Ctx);
+          MCConstantExpr::create(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG, Ctx);
   }
 
   // The MCExpr equivalent of getNumSGPRBlocks/getNumVGPRBlocks:
@@ -6997,7 +6997,9 @@ bool AMDGPUAsmParser::ParseDirectiveAMDGPULDS() {
   if (getParser().parseComma())
     return true;
 
-  unsigned LocalMemorySize = AMDGPU::IsaInfo::getLocalMemorySize(getSTI());
+  unsigned LocalMemorySize =
+      AMDGPU::getLocalMemorySize(AMDGPU::parseArchAMDGCN(getSTI().getCPU()),
+                                 AMDGPU::isFullSIMDMode(getSTI()));
 
   int64_t Size;
   SMLoc SizeLoc = getLoc();
diff --git a/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp b/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
index 3b200d412eb0d..1a5223e3cffab 100644
--- a/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
+++ b/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
@@ -60,7 +60,8 @@ AMDGPUDisassembler::AMDGPUDisassembler(const MCSubtargetInfo &STI,
       MAI(Ctx.getAsmInfo()),
       HwModeRegClass(STI.getHwMode(MCSubtargetInfo::HwMode_RegInfo)),
       TargetMaxInstBytes(MAI.getMaxInstLength(&STI)),
-      TargetID(AMDGPU::createAMDGPUTargetID(STI, "")),
+      TargetID(AMDGPU::TargetID::createFromSubtargetFeatures(
+          STI.getTargetTriple(), STI.getCPU(), "")),
       CodeObjectVersion(AMDGPU::getDefaultAMDHSACodeObjectVersion()) {
   // ToDo: AMDGPUDisassembler supports only VI ISA.
   if (!STI.hasFeature(AMDGPU::FeatureGCN3Encoding) && !isGFX10Plus())
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
index 34d987cd6d266..5c0bc79e8dbd4 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
@@ -223,7 +223,7 @@ GCNSubtarget::GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS,
     : // clang-format off
     AMDGPUGenSubtargetInfo(TT, GPU, /*TuneCPU*/ GPU, FS),
     AMDGPUSubtarget(TT),
-    TargetID(AMDGPU::createAMDGPUTargetID(*this, "")),
+    TargetID(AMDGPU::TargetID::createFromSubtargetFeatures(TT, GPU, "")),
     InstrItins(getInstrItineraryForCPU(GPU)),
     BufferOOBRelaxed(BufferOOBRelaxed),
     TBufferOOBRelaxed(TBufferOOBRelaxed),
@@ -576,7 +576,7 @@ unsigned GCNSubtarget::getBaseMaxNumSGPRs(
   }
 
   if (hasSGPRInitBug())
-    MaxNumSGPRs = AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG;
+    MaxNumSGPRs = AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG;
 
   return std::min(MaxNumSGPRs - ReservedNumSGPRs, MaxAddressableNumSGPRs);
 }
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 2e09f1f1fda39..c8020be9475c6 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -864,7 +864,12 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
   /// \returns Addressable number of architectural VGPRs supported by the
   /// subtarget.
   unsigned getAddressableNumArchVGPRs() const {
-    return AMDGPU::IsaInfo::getAddressableNumArchVGPRs(*this);
+    // The TargetParser query includes AGPRs on targets with a unified register
+    // file, but only 256 registers are architectural VGPRs.
+    if (hasGFX90AInsts())
+      return 256;
+    return AMDGPU::getAddressableNumVGPRs(getTargetID().getGPUKind(),
+                                          isWave32());
   }
 
   /// \returns Addressable number of VGPRs supported by the subtarget.
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
index 6b1e58b52f305..68490fc64f6e0 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
@@ -51,8 +51,9 @@ void AMDGPUTargetStreamer::initializeTargetID(const MCSubtargetInfo &STI,
   assert(TargetID == std::nullopt && "TargetID can only be initialized once");
   // Apply xnack/sramecc from subtarget features only in MC contexts
   // (assembler), not in codegen where they come from module flags
-  TargetID = AMDGPU::createAMDGPUTargetID(
-      STI, ApplyFeatureString ? STI.getFeatureString() : "");
+  TargetID = AMDGPU::TargetID::createFromSubtargetFeatures(
+      STI.getTargetTriple(), STI.getCPU(),
+      ApplyFeatureString ? STI.getFeatureString() : "");
 }
 
 bool AMDGPUTargetStreamer::EmitHSAMetadataV3(StringRef HSAMetadataString) {
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index e45359719d6c9..c56d6599d54ac 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1088,15 +1088,6 @@ VOPD::InstInfo getVOPDInstInfo(unsigned VOPDOpcode,
   return VOPD::InstInfo(OpXInfo, OpYInfo);
 }
 
-TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI,
-                              StringRef FeatureString) {
-  // In codegen the mode comes from module flags and FeatureString is empty, so
-  // the processor defaults apply. The assembler has no target directive, so it
-  // pins the mode via the +xnack/-xnack/+sramecc/-sramecc feature string.
-  return TargetID::createFromSubtargetFeatures(STI.getTargetTriple(),
-                                               STI.getCPU(), FeatureString);
-}
-
 namespace IsaInfo {
 
 unsigned getInstCacheLineSize(const MCSubtargetInfo &STI) {
@@ -1116,57 +1107,6 @@ unsigned getWavefrontSize(const MCSubtargetInfo &STI) {
   return 64;
 }
 
-// Maximum LDS a single work-group can address. This is a fixed HW cap. It does
-// not depend on how many SIMDs a work-group runs on.
-static unsigned getMaxHWAddressableLocalMemorySize(const MCSubtargetInfo &STI) {
-  if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
-    return 32768;
-  if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
-    return 65536;
-  if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
-    return 163840;
-  if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
-    return 196608;
-  if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
-    return 327680;
-  return 32768;
-}
-
-// Total physical size of LDS on the block, in bytes. On targets with
-// FeatureHalfAddressablePhysicalLocalMemory the physical block is twice the
-// addressable size (gfx6: 64 KiB physical and 32 KiB addressable;
-// gfx10/11/12: 128 KiB physical and 64 KiB addressable). On other targets it is
-// equal to the addressable size.
-static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI) {
-  unsigned Addressable = getMaxHWAddressableLocalMemorySize(STI);
-  if (STI.getFeatureBits().test(FeatureHalfAddressablePhysicalLocalMemory))
-    return 2 * Addressable;
-  return Addressable;
-}
-
-// Sizes in use, by generation (addressable / physical block):
-//   gfx6              :  32 KiB addressable, 64 KiB physical block
-//   gfx7 / gfx8 / gfx9:  64 KiB
-//   gfx9.5 (gfx950)   : 160 KiB
-//   gfx10 / 11 / 12   :  64 KiB addressable, 128 KiB physical block
-//   gfx12.5 (gfx1250) : 320 KiB (always runs on four SIMDs)
-//   gfx13             : 192 KiB on four SIMDs, 96 KiB on two
-// Total available in the current mode. The physical size is halved when a
-// work-group runs on two SIMDs.
-unsigned getLocalMemorySize(const MCSubtargetInfo &STI) {
-  unsigned Size = getPhysicalLocalMemorySize(STI);
-  if (!isFullSIMDMode(STI))
-    Size /= 2;
-  return Size;
-}
-
-// What one work-group can allocate in the current mode. This is the HW
-// addressable cap, but never more than the total available in the current mode.
-unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI) {
-  return std::min(getMaxHWAddressableLocalMemorySize(STI),
-                  getLocalMemorySize(STI));
-}
-
 unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI,
                                unsigned FlatWorkGroupSize) {
   assert(FlatWorkGroupSize != 0);
@@ -1301,23 +1241,15 @@ unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs) {
 unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI,
                              unsigned DynamicVGPRBlockSize,
                              std::optional<bool> EnableWavefrontSize32) {
-  if (STI.getFeatureBits().test(FeatureGFX90AInsts))
-    return 8;
-
-  if (DynamicVGPRBlockSize != 0)
+  if (DynamicVGPRBlockSize != 0 &&
+      !STI.getFeatureBits().test(FeatureGFX90AInsts))
     return DynamicVGPRBlockSize;
 
   bool IsWave32 = EnableWavefrontSize32
                       ? *EnableWavefrontSize32
                       : STI.getFeatureBits().test(FeatureWavefrontSize32);
 
-  if (STI.getFeatureBits().test(Feature1536VGPRs))
-    return IsWave32 ? 24 : 12;
-
-  if (hasGFX10_3Insts(STI))
-    return IsWave32 ? 16 : 8;
-
-  return IsWave32 ? 8 : 4;
+  return AMDGPU::getVGPRAllocGranule(parseArchAMDGCN(STI.getCPU()), IsWave32);
 }
 
 unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI,
@@ -1337,25 +1269,16 @@ unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI,
 
 unsigned getArchVGPRAllocGranule() { return 4; }
 
-unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI) {
-  const auto &Features = STI.getFeatureBits();
-  if (Features.test(Feature1024AddressableVGPRs))
-    return Features.test(FeatureWavefrontSize32) ? 1024 : 512;
-  return 256;
-}
-
 unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
                                 unsigned DynamicVGPRBlockSize) {
-  const auto &Features = STI.getFeatureBits();
-  if (Features.test(FeatureGFX90AInsts))
-    return 512;
-
-  if (DynamicVGPRBlockSize != 0) {
+  if (DynamicVGPRBlockSize != 0 &&
+      !STI.getFeatureBits().test(FeatureGFX90AInsts)) {
     // On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
-    return MaxDynamicVGPRBlocks *
-           getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
+    return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
   }
-  return getAddressableNumArchVGPRs(STI);
+  return AMDGPU::getAddressableNumVGPRs(
+      parseArchAMDGCN(STI.getCPU()),
+      STI.getFeatureBits().test(FeatureWavefrontSize32));
 }
 
 unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI,
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index bce059a0c18a7..b4a37c6fe4be4 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -161,20 +161,9 @@ struct WMMAInstInfo {
 #define GET_WMMAInstInfoTable_DECL
 #include "AMDGPUGenSearchableTables.inc"
 
-using TargetIDSetting = AMDGPU::TargetIDSetting;
-using TargetID = AMDGPU::TargetID;
-
-/// Construct TargetID from MCSubtargetInfo. \p FeatureString is used to
-/// determine explicitly requested xnack/sramecc settings.
-TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI,
-                              StringRef FeatureString);
-
 namespace IsaInfo {
 
-enum {
-  FIXED_NUM_SGPRS_FOR_INIT_BUG = AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG,
-  TRAP_NUM_SGPRS = 16
-};
+enum { TRAP_NUM_SGPRS = 16 };
 
 /// Returns true if \p Lhs and \p Rhs are incompatible (both specific but
 /// different).
@@ -189,13 +178,6 @@ unsigned getInstCacheLineSize(const MCSubtargetInfo &STI);
 /// \returns Wavefront size for given subtarget \p STI.
 unsigned getWavefrontSize(const MCSubtargetInfo &STI);
 
-/// \returns Local memory size in bytes for given subtarget \p STI.
-unsigned getLocalMemorySize(const MCSubtargetInfo &STI);
-
-/// \returns Maximum addressable local memory size in bytes for given subtarget
-/// \p STI.
-unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI);
-
 /// \returns Maximum number of work groups per compute unit for given subtarget
 /// \p STI and limited by given \p FlatWorkGroupSize.
 unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI,
@@ -256,10 +238,6 @@ unsigned getArchVGPRAllocGranule();
 /// Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
 static constexpr unsigned MaxDynamicVGPRBlocks = 8;
 
-/// \returns Addressable number of architectural VGPRs for a given subtarget \p
-/// STI.
-unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI);
-
 /// \returns Addressable number of VGPRs for given subtarget \p STI.
 unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
                                 unsigned DynamicVGPRBlockSize);
diff --git a/llvm/test/MC/AMDGPU/elf-lds-size.s b/llvm/test/MC/AMDGPU/elf-lds-size.s
new file mode 100644
index 0000000000000..88b45e75316e6
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/elf-lds-size.s
@@ -0,0 +1,15 @@
+// RUN: not llvm-mc -triple=amdgpu6.00-- -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu9.00-- -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu9.50-- -defsym LDS_SIZE=163840 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu10.30-- -mattr=-cumode -defsym LDS_SIZE=131072 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu10.30-- -mattr=+cumode -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu12.50-- -mattr=-cumode -defsym LDS_SIZE=327680 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu12.50-- -mattr=+cumode -defsym LDS_SIZE=327680 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu13.10-- -mattr=-cumode -defsym LDS_SIZE=196608 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu13.10-- -mattr=+cumode -defsym LDS_SIZE=98304 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+
+// LDS symbols can use the physical LDS available in the current mode, even
+// when it is larger than the amount a single work-group can address.
+.amdgpu_lds at_limit, LDS_SIZE
+.amdgpu_lds over_limit, LDS_SIZE + 1
+// CHECK: :[[@LINE-1]]:25: error: size is too large
diff --git a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
index a21165d041bc6..2994fb2c3ceb2 100644
--- a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
+++ b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
@@ -216,8 +216,8 @@ int main(int argc, char **argv) {
   unsigned WaveSize = AMDGPU::IsaInfo::getWavefrontSize(STI);
   unsigned MaxWaves = ST.getMaxWavesPerEU();
   unsigned NumWorkGroupSIMDs = ST.getNumWorkGroupSIMDs();
-  unsigned LocalMemSize = AMDGPU::IsaInfo::getLocalMemorySize(STI);
-  unsigned AddrLocalMem = AMDGPU::IsaInfo::getAddressableLocalMemorySize(STI);
+  unsigned LocalMemSize = ST.getLocalMemorySize();
+  unsigned AddrLocalMem = ST.getAddressableLocalMemorySize();
   unsigned AddrVGPRs =
       AMDGPU::IsaInfo::getAddressableNumVGPRs(STI, DynVGPRBlockSize);
   unsigned AddrSGPRs = ST.getAddressableNumSGPRs();



More information about the llvm-commits mailing list