[llvm] [AMDGPU] Remove BaseInfo duplicates of TargetParser APIs (PR #226744)

via llvm-commits llvm-commits at lists.llvm.org
Sat Sep 26 19:27:29 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-backend-amdgpu

Author: Chinmay Deshpande (chinmaydd)

<details>
<summary>Changes</summary>

Use TargetParser for LDS sizes and VGPR hardware limits, retaining the backend handling of dynamic VGPR allocation and architectural VGPRs on targets with a unified register file.

---
Full diff: https://github.com/llvm/llvm-project/pull/226744.diff


10 Files Affected:

- (modified) llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp (+2-3) 
- (modified) llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp (+4-2) 
- (modified) llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp (+2-1) 
- (modified) llvm/lib/Target/AMDGPU/GCNSubtarget.cpp (+2-2) 
- (modified) llvm/lib/Target/AMDGPU/GCNSubtarget.h (+6-1) 
- (modified) llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp (+3-2) 
- (modified) llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp (+9-86) 
- (modified) llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h (+1-23) 
- (added) llvm/test/MC/AMDGPU/elf-lds-size.s (+15) 
- (modified) llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp (+2-2) 


``````````diff
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index c18b840c36791..7650d8b5bf15f 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1390,10 +1390,9 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
   }
 
   if (STM.hasSGPRInitBug()) {
-    ProgInfo.NumSGPR =
-        CreateExpr(AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG);
+    ProgInfo.NumSGPR = CreateExpr(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG);
     ProgInfo.NumSGPRsForWavesPerEU =
-        CreateExpr(AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG);
+        CreateExpr(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG);
   }
 
   if (MFI->getNumUserSGPRs() > STM.getMaxNumUserSGPRs()) {
diff --git a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
index 812acc8f8be85..88d5f5de25312 100644
--- a/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
+++ b/llvm/lib/Target/AMDGPU/AsmParser/AMDGPUAsmParser.cpp
@@ -6175,7 +6175,7 @@ bool AMDGPUAsmParser::calculateGPRBlocks(
 
     if (Features.test(FeatureSGPRInitBug))
       NumSGPRs =
-          MCConstantExpr::create(IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG, Ctx);
+          MCConstantExpr::create(AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG, Ctx);
   }
 
   // The MCExpr equivalent of getNumSGPRBlocks/getNumVGPRBlocks:
@@ -6997,7 +6997,9 @@ bool AMDGPUAsmParser::ParseDirectiveAMDGPULDS() {
   if (getParser().parseComma())
     return true;
 
-  unsigned LocalMemorySize = AMDGPU::IsaInfo::getLocalMemorySize(getSTI());
+  unsigned LocalMemorySize =
+      AMDGPU::getLocalMemorySize(AMDGPU::parseArchAMDGCN(getSTI().getCPU()),
+                                 AMDGPU::isFullSIMDMode(getSTI()));
 
   int64_t Size;
   SMLoc SizeLoc = getLoc();
diff --git a/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp b/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
index 3b200d412eb0d..1a5223e3cffab 100644
--- a/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
+++ b/llvm/lib/Target/AMDGPU/Disassembler/AMDGPUDisassembler.cpp
@@ -60,7 +60,8 @@ AMDGPUDisassembler::AMDGPUDisassembler(const MCSubtargetInfo &STI,
       MAI(Ctx.getAsmInfo()),
       HwModeRegClass(STI.getHwMode(MCSubtargetInfo::HwMode_RegInfo)),
       TargetMaxInstBytes(MAI.getMaxInstLength(&STI)),
-      TargetID(AMDGPU::createAMDGPUTargetID(STI, "")),
+      TargetID(AMDGPU::TargetID::createFromSubtargetFeatures(
+          STI.getTargetTriple(), STI.getCPU(), "")),
       CodeObjectVersion(AMDGPU::getDefaultAMDHSACodeObjectVersion()) {
   // ToDo: AMDGPUDisassembler supports only VI ISA.
   if (!STI.hasFeature(AMDGPU::FeatureGCN3Encoding) && !isGFX10Plus())
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
index 34d987cd6d266..5c0bc79e8dbd4 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
@@ -223,7 +223,7 @@ GCNSubtarget::GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS,
     : // clang-format off
     AMDGPUGenSubtargetInfo(TT, GPU, /*TuneCPU*/ GPU, FS),
     AMDGPUSubtarget(TT),
-    TargetID(AMDGPU::createAMDGPUTargetID(*this, "")),
+    TargetID(AMDGPU::TargetID::createFromSubtargetFeatures(TT, GPU, "")),
     InstrItins(getInstrItineraryForCPU(GPU)),
     BufferOOBRelaxed(BufferOOBRelaxed),
     TBufferOOBRelaxed(TBufferOOBRelaxed),
@@ -576,7 +576,7 @@ unsigned GCNSubtarget::getBaseMaxNumSGPRs(
   }
 
   if (hasSGPRInitBug())
-    MaxNumSGPRs = AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG;
+    MaxNumSGPRs = AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG;
 
   return std::min(MaxNumSGPRs - ReservedNumSGPRs, MaxAddressableNumSGPRs);
 }
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index 2e09f1f1fda39..c8020be9475c6 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -864,7 +864,12 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
   /// \returns Addressable number of architectural VGPRs supported by the
   /// subtarget.
   unsigned getAddressableNumArchVGPRs() const {
-    return AMDGPU::IsaInfo::getAddressableNumArchVGPRs(*this);
+    // The TargetParser query includes AGPRs on targets with a unified register
+    // file, but only 256 registers are architectural VGPRs.
+    if (hasGFX90AInsts())
+      return 256;
+    return AMDGPU::getAddressableNumVGPRs(getTargetID().getGPUKind(),
+                                          isWave32());
   }
 
   /// \returns Addressable number of VGPRs supported by the subtarget.
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
index 6b1e58b52f305..68490fc64f6e0 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUTargetStreamer.cpp
@@ -51,8 +51,9 @@ void AMDGPUTargetStreamer::initializeTargetID(const MCSubtargetInfo &STI,
   assert(TargetID == std::nullopt && "TargetID can only be initialized once");
   // Apply xnack/sramecc from subtarget features only in MC contexts
   // (assembler), not in codegen where they come from module flags
-  TargetID = AMDGPU::createAMDGPUTargetID(
-      STI, ApplyFeatureString ? STI.getFeatureString() : "");
+  TargetID = AMDGPU::TargetID::createFromSubtargetFeatures(
+      STI.getTargetTriple(), STI.getCPU(),
+      ApplyFeatureString ? STI.getFeatureString() : "");
 }
 
 bool AMDGPUTargetStreamer::EmitHSAMetadataV3(StringRef HSAMetadataString) {
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index e45359719d6c9..c56d6599d54ac 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1088,15 +1088,6 @@ VOPD::InstInfo getVOPDInstInfo(unsigned VOPDOpcode,
   return VOPD::InstInfo(OpXInfo, OpYInfo);
 }
 
-TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI,
-                              StringRef FeatureString) {
-  // In codegen the mode comes from module flags and FeatureString is empty, so
-  // the processor defaults apply. The assembler has no target directive, so it
-  // pins the mode via the +xnack/-xnack/+sramecc/-sramecc feature string.
-  return TargetID::createFromSubtargetFeatures(STI.getTargetTriple(),
-                                               STI.getCPU(), FeatureString);
-}
-
 namespace IsaInfo {
 
 unsigned getInstCacheLineSize(const MCSubtargetInfo &STI) {
@@ -1116,57 +1107,6 @@ unsigned getWavefrontSize(const MCSubtargetInfo &STI) {
   return 64;
 }
 
-// Maximum LDS a single work-group can address. This is a fixed HW cap. It does
-// not depend on how many SIMDs a work-group runs on.
-static unsigned getMaxHWAddressableLocalMemorySize(const MCSubtargetInfo &STI) {
-  if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
-    return 32768;
-  if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
-    return 65536;
-  if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
-    return 163840;
-  if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
-    return 196608;
-  if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
-    return 327680;
-  return 32768;
-}
-
-// Total physical size of LDS on the block, in bytes. On targets with
-// FeatureHalfAddressablePhysicalLocalMemory the physical block is twice the
-// addressable size (gfx6: 64 KiB physical and 32 KiB addressable;
-// gfx10/11/12: 128 KiB physical and 64 KiB addressable). On other targets it is
-// equal to the addressable size.
-static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI) {
-  unsigned Addressable = getMaxHWAddressableLocalMemorySize(STI);
-  if (STI.getFeatureBits().test(FeatureHalfAddressablePhysicalLocalMemory))
-    return 2 * Addressable;
-  return Addressable;
-}
-
-// Sizes in use, by generation (addressable / physical block):
-//   gfx6              :  32 KiB addressable, 64 KiB physical block
-//   gfx7 / gfx8 / gfx9:  64 KiB
-//   gfx9.5 (gfx950)   : 160 KiB
-//   gfx10 / 11 / 12   :  64 KiB addressable, 128 KiB physical block
-//   gfx12.5 (gfx1250) : 320 KiB (always runs on four SIMDs)
-//   gfx13             : 192 KiB on four SIMDs, 96 KiB on two
-// Total available in the current mode. The physical size is halved when a
-// work-group runs on two SIMDs.
-unsigned getLocalMemorySize(const MCSubtargetInfo &STI) {
-  unsigned Size = getPhysicalLocalMemorySize(STI);
-  if (!isFullSIMDMode(STI))
-    Size /= 2;
-  return Size;
-}
-
-// What one work-group can allocate in the current mode. This is the HW
-// addressable cap, but never more than the total available in the current mode.
-unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI) {
-  return std::min(getMaxHWAddressableLocalMemorySize(STI),
-                  getLocalMemorySize(STI));
-}
-
 unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI,
                                unsigned FlatWorkGroupSize) {
   assert(FlatWorkGroupSize != 0);
@@ -1301,23 +1241,15 @@ unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs) {
 unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI,
                              unsigned DynamicVGPRBlockSize,
                              std::optional<bool> EnableWavefrontSize32) {
-  if (STI.getFeatureBits().test(FeatureGFX90AInsts))
-    return 8;
-
-  if (DynamicVGPRBlockSize != 0)
+  if (DynamicVGPRBlockSize != 0 &&
+      !STI.getFeatureBits().test(FeatureGFX90AInsts))
     return DynamicVGPRBlockSize;
 
   bool IsWave32 = EnableWavefrontSize32
                       ? *EnableWavefrontSize32
                       : STI.getFeatureBits().test(FeatureWavefrontSize32);
 
-  if (STI.getFeatureBits().test(Feature1536VGPRs))
-    return IsWave32 ? 24 : 12;
-
-  if (hasGFX10_3Insts(STI))
-    return IsWave32 ? 16 : 8;
-
-  return IsWave32 ? 8 : 4;
+  return AMDGPU::getVGPRAllocGranule(parseArchAMDGCN(STI.getCPU()), IsWave32);
 }
 
 unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI,
@@ -1337,25 +1269,16 @@ unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI,
 
 unsigned getArchVGPRAllocGranule() { return 4; }
 
-unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI) {
-  const auto &Features = STI.getFeatureBits();
-  if (Features.test(Feature1024AddressableVGPRs))
-    return Features.test(FeatureWavefrontSize32) ? 1024 : 512;
-  return 256;
-}
-
 unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
                                 unsigned DynamicVGPRBlockSize) {
-  const auto &Features = STI.getFeatureBits();
-  if (Features.test(FeatureGFX90AInsts))
-    return 512;
-
-  if (DynamicVGPRBlockSize != 0) {
+  if (DynamicVGPRBlockSize != 0 &&
+      !STI.getFeatureBits().test(FeatureGFX90AInsts)) {
     // On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
-    return MaxDynamicVGPRBlocks *
-           getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
+    return MaxDynamicVGPRBlocks * DynamicVGPRBlockSize;
   }
-  return getAddressableNumArchVGPRs(STI);
+  return AMDGPU::getAddressableNumVGPRs(
+      parseArchAMDGCN(STI.getCPU()),
+      STI.getFeatureBits().test(FeatureWavefrontSize32));
 }
 
 unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI,
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index bce059a0c18a7..b4a37c6fe4be4 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -161,20 +161,9 @@ struct WMMAInstInfo {
 #define GET_WMMAInstInfoTable_DECL
 #include "AMDGPUGenSearchableTables.inc"
 
-using TargetIDSetting = AMDGPU::TargetIDSetting;
-using TargetID = AMDGPU::TargetID;
-
-/// Construct TargetID from MCSubtargetInfo. \p FeatureString is used to
-/// determine explicitly requested xnack/sramecc settings.
-TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI,
-                              StringRef FeatureString);
-
 namespace IsaInfo {
 
-enum {
-  FIXED_NUM_SGPRS_FOR_INIT_BUG = AMDGPU::FIXED_NUM_SGPRS_FOR_INIT_BUG,
-  TRAP_NUM_SGPRS = 16
-};
+enum { TRAP_NUM_SGPRS = 16 };
 
 /// Returns true if \p Lhs and \p Rhs are incompatible (both specific but
 /// different).
@@ -189,13 +178,6 @@ unsigned getInstCacheLineSize(const MCSubtargetInfo &STI);
 /// \returns Wavefront size for given subtarget \p STI.
 unsigned getWavefrontSize(const MCSubtargetInfo &STI);
 
-/// \returns Local memory size in bytes for given subtarget \p STI.
-unsigned getLocalMemorySize(const MCSubtargetInfo &STI);
-
-/// \returns Maximum addressable local memory size in bytes for given subtarget
-/// \p STI.
-unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI);
-
 /// \returns Maximum number of work groups per compute unit for given subtarget
 /// \p STI and limited by given \p FlatWorkGroupSize.
 unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI,
@@ -256,10 +238,6 @@ unsigned getArchVGPRAllocGranule();
 /// Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
 static constexpr unsigned MaxDynamicVGPRBlocks = 8;
 
-/// \returns Addressable number of architectural VGPRs for a given subtarget \p
-/// STI.
-unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI);
-
 /// \returns Addressable number of VGPRs for given subtarget \p STI.
 unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI,
                                 unsigned DynamicVGPRBlockSize);
diff --git a/llvm/test/MC/AMDGPU/elf-lds-size.s b/llvm/test/MC/AMDGPU/elf-lds-size.s
new file mode 100644
index 0000000000000..88b45e75316e6
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/elf-lds-size.s
@@ -0,0 +1,15 @@
+// RUN: not llvm-mc -triple=amdgpu6.00-- -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu9.00-- -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu9.50-- -defsym LDS_SIZE=163840 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu10.30-- -mattr=-cumode -defsym LDS_SIZE=131072 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu10.30-- -mattr=+cumode -defsym LDS_SIZE=65536 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu12.50-- -mattr=-cumode -defsym LDS_SIZE=327680 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu12.50-- -mattr=+cumode -defsym LDS_SIZE=327680 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu13.10-- -mattr=-cumode -defsym LDS_SIZE=196608 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+// RUN: not llvm-mc -triple=amdgpu13.10-- -mattr=+cumode -defsym LDS_SIZE=98304 -filetype=null %s 2>&1 | FileCheck %s --implicit-check-not=error:
+
+// LDS symbols can use the physical LDS available in the current mode, even
+// when it is larger than the amount a single work-group can address.
+.amdgpu_lds at_limit, LDS_SIZE
+.amdgpu_lds over_limit, LDS_SIZE + 1
+// CHECK: :[[@LINE-1]]:25: error: size is too large
diff --git a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
index a21165d041bc6..2994fb2c3ceb2 100644
--- a/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
+++ b/llvm/tools/llvm-calc-occupancy/llvm-calc-occupancy.cpp
@@ -216,8 +216,8 @@ int main(int argc, char **argv) {
   unsigned WaveSize = AMDGPU::IsaInfo::getWavefrontSize(STI);
   unsigned MaxWaves = ST.getMaxWavesPerEU();
   unsigned NumWorkGroupSIMDs = ST.getNumWorkGroupSIMDs();
-  unsigned LocalMemSize = AMDGPU::IsaInfo::getLocalMemorySize(STI);
-  unsigned AddrLocalMem = AMDGPU::IsaInfo::getAddressableLocalMemorySize(STI);
+  unsigned LocalMemSize = ST.getLocalMemorySize();
+  unsigned AddrLocalMem = ST.getAddressableLocalMemorySize();
   unsigned AddrVGPRs =
       AMDGPU::IsaInfo::getAddressableNumVGPRs(STI, DynVGPRBlockSize);
   unsigned AddrSGPRs = ST.getAddressableNumSGPRs();

``````````

</details>


https://github.com/llvm/llvm-project/pull/226744


More information about the llvm-commits mailing list