[llvm] [AMDGPU] Add `getLDSEncodingGranule` to TargetParser (PR #224852)

Chinmay Deshpande via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 11:58:42 PDT 2026


https://github.com/chinmaydd updated https://github.com/llvm/llvm-project/pull/224852

>From 6959442887f9e795b4b745070a216759111ce17f Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sat, 19 Sep 2026 14:15:35 -0400
Subject: [PATCH 1/3] [AMDGPU] Add LDS encoding granularity to TargetParser

Model LDS encoding granularity with dedicated features and expose the
byte-valued getLDSEncodingGranule query for GPUKind and subarch. Keep
encoding independent of the hardware allocation granularity used for
occupancy; GFX10.3, GFX11 and GFX12.0 encode in 512-byte units while
allocating 1024-byte blocks.

Migrate program resource register and PAL metadata encoding to the new
query and remove getLdsDwGranularity from AMDGPUBaseInfo. gfx9-4-generic
uses gfx950's 1280-byte encoding granule independently of LDS capacity.

Test encoding queries, feature membership, generic-target validation and
encoded LDS sizes, including the GFX10.3 allocation/encoding distinction.

Change-Id: I9d3c2c041605e9a45fa8fbda09fc3460a74953ea
---
 .../llvm/TargetParser/AMDGPUTargetParser.h    |  5 ++
 llvm/lib/Target/AMDGPU/AMDGPU.td              | 20 +++++
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp   |  7 +-
 llvm/lib/Target/AMDGPU/AMDGPUFeatures.td      | 15 ++++
 llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h      |  1 +
 llvm/lib/Target/AMDGPU/R600Processors.td      |  2 +
 .../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp    | 14 ---
 llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h |  5 --
 llvm/lib/TargetParser/AMDGPUTargetParser.cpp  | 35 +++++++-
 llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll  | 26 ++++++
 .../CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll | 34 +++++++
 .../AMDGPUTargetDefLDSAllocGranularity.td     | 42 +++++++--
 .../TargetParser/TargetParserTest.cpp         | 88 ++++++++++++-------
 13 files changed, 235 insertions(+), 59 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll
 create mode 100644 llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll

diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index 49245a4322ab6..1239baa200a73 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -245,6 +245,11 @@ LLVM_ABI unsigned getLDSBankCount(Triple::SubArchType SubArch);
 LLVM_ABI unsigned getLDSAllocGranule(GPUKind AK);
 LLVM_ABI unsigned getLDSAllocGranule(Triple::SubArchType SubArch);
 
+/// \returns LDS size encoding granularity in bytes, used for program resource
+/// registers and metadata. This can differ from the allocation granularity.
+LLVM_ABI unsigned getLDSEncodingGranule(GPUKind AK);
+LLVM_ABI unsigned getLDSEncodingGranule(Triple::SubArchType SubArch);
+
 /// \returns Number of SIMDs a work-group's waves run on. All four SIMDs of the
 /// functional block in full-SIMD mode, half of them otherwise.
 constexpr unsigned getNumWorkGroupSIMDs(bool FullSIMDMode) {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index 186191d76cb38..8ddbd8a44707a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -1654,6 +1654,7 @@ def FeatureSouthernIslands : GCNSubtargetFeatureGeneration<"SOUTHERN_ISLANDS",
   [FeatureFP64, FeatureAddressableLocalMemorySize32768,
   FeatureHalfAddressablePhysicalLocalMemory,
   FeatureLDSAllocGranularity256,
+  FeatureLDSEncodingGranularity256,
   FeatureMIMG_R128,
   FeatureWavefrontSize64, FeatureSupportsWave64, FeatureSMemTimeInst,
   FeatureMadMacF32Insts,
@@ -1674,6 +1675,7 @@ def FeatureSeaIslands : GCNSubtargetFeatureGeneration<"SEA_ISLANDS",
     "sea-islands",
   [FeatureFP64, FeatureAddressableLocalMemorySize65536,
   FeatureLDSAllocGranularity512,
+  FeatureLDSEncodingGranularity512,
   FeatureMIMG_R128,
   FeatureWavefrontSize64, FeatureSupportsWave64, FeatureFlatAddressSpace,
   FeatureCIInsts, FeatureMovrel, FeatureTrigReducedRange,
@@ -1696,6 +1698,7 @@ def FeatureVolcanicIslands : GCNSubtargetFeatureGeneration<"VOLCANIC_ISLANDS",
   "volcanic-islands",
   [FeatureFP64, FeatureAddressableLocalMemorySize65536,
    FeatureLDSAllocGranularity512,
+   FeatureLDSEncodingGranularity512,
    FeatureMIMG_R128,
    FeatureWavefrontSize64, FeatureSupportsWave64, FeatureFlatAddressSpace,
    FeatureGCN3Encoding, FeatureCIInsts, Feature16BitInsts,
@@ -1746,6 +1749,7 @@ def FeatureGFX9 : GCNSubtargetFeatureGeneration<"GFX9",
 def FeatureGFX10 : GCNSubtargetFeatureGeneration<"GFX10",
   "gfx10",
   [FeatureFP64, FeatureAddressableLocalMemorySize65536,
+   FeatureLDSEncodingGranularity512,
    FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128,
    FeatureSupportsWave32, FeatureSupportsWave64, FeatureSupportsWGP,
    FeatureFlatAddressSpace,
@@ -1781,6 +1785,7 @@ def FeatureGFX11 : GCNSubtargetFeatureGeneration<"GFX11",
   "gfx11",
   [FeatureFP64, FeatureAddressableLocalMemorySize65536,
    FeatureLDSAllocGranularity1024,
+   FeatureLDSEncodingGranularity512,
    FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128,
    FeatureSupportsWave32, FeatureSupportsWave64, FeatureSupportsWGP,
    FeatureFlatAddressSpace, Feature16BitInsts,
@@ -1951,6 +1956,7 @@ def FeatureISAVersion9_0_Common : FeatureSet<
   [FeatureGFX9,
    FeatureAddressableLocalMemorySize65536,
    FeatureLDSAllocGranularity512,
+   FeatureLDSEncodingGranularity512,
    FeatureLDSBankCount32,
    FeatureImageInsts,
    FeatureMadMacF32Insts]>;
@@ -2101,6 +2107,7 @@ def FeatureISAVersion9_5_Common : FeatureSet<
   !listconcat(FeatureISAVersion9_4_Common.Features,
   [FeatureAddressableLocalMemorySize163840,
    FeatureLDSAllocGranularity1280,
+   FeatureLDSEncodingGranularity1280,
    FeatureLDSBankCount64,
    FeatureFP8Insts,
    FeatureFP8ConversionInsts,
@@ -2124,6 +2131,7 @@ def FeatureISAVersion9_4_2 : FeatureSet<
     [
       FeatureAddressableLocalMemorySize65536,
       FeatureLDSAllocGranularity512,
+      FeatureLDSEncodingGranularity512,
       FeatureLDSBankCount32,
       FeatureFP8Insts,
       FeatureFP8ConversionInsts,
@@ -2136,6 +2144,7 @@ def FeatureISAVersion9_4_Generic : FeatureSet<
   !listconcat(FeatureISAVersion9_4_Common.Features,
     [FeatureAddressableLocalMemorySize65536,
      FeatureLDSAllocGranularity1280,
+     FeatureLDSEncodingGranularity1280,
      FeatureLDSBankCount32,
      FeatureRequiresCOV6])>;
 
@@ -2348,6 +2357,7 @@ def FeatureISAVersion12 : FeatureSet<
    FeatureBackOffBarrier,
    FeatureAddressableLocalMemorySize65536,
    FeatureLDSAllocGranularity1024,
+   FeatureLDSEncodingGranularity512,
    FeatureHalfAddressablePhysicalLocalMemory,
    FeatureLDSBankCount32,
    FeatureDLInsts,
@@ -2507,6 +2517,7 @@ def FeatureISAVersion12_50_STRICT : FeatureSet<
   [FeatureGFX1250_STRICT,
    FeatureAddressableLocalMemorySize327680,
    FeatureLDSAllocGranularity2048,
+   FeatureLDSEncodingGranularity2048,
    FeatureLDSBankCount64,
    FeatureWMMAN16Insts,
    FeatureVOP3PX2IncrementsVaVdstTwice,
@@ -2538,6 +2549,7 @@ def FeatureISAVersion12_50 : FeatureSet<
   !listconcat(FeatureISAVersion12_50_Common.Features,
   [FeatureAddressableLocalMemorySize327680,
    FeatureLDSAllocGranularity2048,
+   FeatureLDSEncodingGranularity2048,
    FeatureLDSBankCount64,
    FeatureWMMAN16Insts,
    FeatureWMMAF4Insts,
@@ -2570,6 +2582,7 @@ def FeatureISAVersion12_51 : FeatureSet<
   !listconcat(FeatureISAVersion12_50_Common.Features,
   [FeatureAddressableLocalMemorySize327680,
    FeatureLDSAllocGranularity2048,
+   FeatureLDSEncodingGranularity2048,
    FeatureLDSBankCount64,
    FeatureWMMAN16Insts,
    FeatureWMMAF4Insts,
@@ -2610,6 +2623,7 @@ def FeatureISAVersion12_5_Generic: FeatureSet<
   !listconcat(FeatureISAVersion12_50_Common.Features,
   [FeatureAddressableLocalMemorySize327680,
    FeatureLDSAllocGranularity2048,
+   FeatureLDSEncodingGranularity2048,
    FeatureLDSBankCount64,
    FeatureBlock16ConversionScaleInsts,
    FeatureSetregVGPRMSBFixup,
@@ -2629,6 +2643,7 @@ def FeatureISAVersion13 : FeatureSet<
    FeatureWaveMatchInsts,
    FeatureAddressableLocalMemorySize196608,
    FeatureLDSAllocGranularity1024,
+   FeatureLDSEncodingGranularity1024,
    Feature64BitLiterals,
    FeatureLDSBankCount32,
    FeatureDLInsts,
@@ -3323,6 +3338,11 @@ def AMDGPUFrontendVisibleFeatures {
   FeatureLDSAllocGranularity1024,
   FeatureLDSAllocGranularity1280,
   FeatureLDSAllocGranularity2048,
+  FeatureLDSEncodingGranularity256,
+  FeatureLDSEncodingGranularity512,
+  FeatureLDSEncodingGranularity1024,
+  FeatureLDSEncodingGranularity1280,
+  FeatureLDSEncodingGranularity2048,
   ];
 }
 
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 11c482c83f1e1..f25448a81f457 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1440,7 +1440,8 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
 
   ProgInfo.LDSSize = MFI->getLDSSize();
 
-  unsigned LDSGranularityBytes = getLdsDwGranularity(STM) * 4;
+  unsigned LDSGranularityBytes =
+      AMDGPU::getLDSEncodingGranule(STM.getTargetID().getGPUKind());
   ProgInfo.LDSBlocks =
       alignTo(ProgInfo.LDSSize, LDSGranularityBytes) / LDSGranularityBytes;
 
@@ -1679,8 +1680,8 @@ static void EmitPALMetadataCommon(AMDGPUPALMetadata *MD,
 
   MD->updateHwStageMaximum(
       CC, ".lds_size",
-      (unsigned)(CurrentProgramInfo.LdsSize * getLdsDwGranularity(ST) *
-                 sizeof(uint32_t)));
+      (unsigned)(CurrentProgramInfo.LdsSize *
+                 AMDGPU::getLDSEncodingGranule(ST.getTargetID().getGPUKind())));
 }
 
 // This is the equivalent of EmitProgramInfoSI above, but for when the OS type
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td b/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
index 49c90acabb74b..cd3ac06f60fd1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
@@ -60,6 +60,21 @@ def FeatureLDSAllocGranularity1024 : SubtargetFeatureLDSAllocGranularity<1024>;
 def FeatureLDSAllocGranularity1280 : SubtargetFeatureLDSAllocGranularity<1280>;
 def FeatureLDSAllocGranularity2048 : SubtargetFeatureLDSAllocGranularity<2048>;
 
+// Units used to encode LDS size in program resource registers and metadata.
+// These can differ from the hardware allocation granularity. A generic target's
+// encoding granularity must be present on at least one covered GPU.
+class SubtargetFeatureLDSEncodingGranularity <int Granularity> : AMDGPUGenericAnyFeature <
+  "lds-encoding-granularity-"#Granularity,
+  "LDSEncodingGranularity",
+  !cast<string>(Granularity),
+  "LDS encoding granularity in bytes.">;
+
+def FeatureLDSEncodingGranularity256 : SubtargetFeatureLDSEncodingGranularity<256>;
+def FeatureLDSEncodingGranularity512 : SubtargetFeatureLDSEncodingGranularity<512>;
+def FeatureLDSEncodingGranularity1024 : SubtargetFeatureLDSEncodingGranularity<1024>;
+def FeatureLDSEncodingGranularity1280 : SubtargetFeatureLDSEncodingGranularity<1280>;
+def FeatureLDSEncodingGranularity2048 : SubtargetFeatureLDSEncodingGranularity<2048>;
+
 // Whether each wavefront size mode is available on the hardware,
 // independent of the active mode.
 def FeatureSupportsWave32 : SubtargetFeature<"supports-wave32",
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
index e1f331229c463..d2ab61671c932 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
@@ -60,6 +60,7 @@ class AMDGPUSubtarget {
   unsigned LocalMemorySize = 0;
   unsigned AddressableLocalMemorySize = 0;
   unsigned LDSAllocationGranularity = 0;
+  unsigned LDSEncodingGranularity = 0;
   char WavefrontSizeLog2 = 0;
   unsigned FlatOffsetBitWidth = 0;
 
diff --git a/llvm/lib/Target/AMDGPU/R600Processors.td b/llvm/lib/Target/AMDGPU/R600Processors.td
index 04947b4c94b7c..a72c367fd74a2 100644
--- a/llvm/lib/Target/AMDGPU/R600Processors.td
+++ b/llvm/lib/Target/AMDGPU/R600Processors.td
@@ -61,6 +61,7 @@ def FeatureR700 : R600SubtargetFeatureGeneration<"R700", "r700",
 def FeatureEvergreen : R600SubtargetFeatureGeneration<"EVERGREEN", "evergreen",
   [FeatureFetchLimit16, FeatureAddressableLocalMemorySize32768,
    FeatureLDSAllocGranularity256,
+   FeatureLDSEncodingGranularity256,
    FeatureMadMacF32Insts]
 >;
 
@@ -69,6 +70,7 @@ def FeatureNorthernIslands : R600SubtargetFeatureGeneration<"NORTHERN_ISLANDS",
   [FeatureFetchLimit16, FeatureWavefrontSize64,
    FeatureAddressableLocalMemorySize32768,
    FeatureLDSAllocGranularity256,
+   FeatureLDSEncodingGranularity256,
    FeatureMadMacF32Insts]
 >;
 
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 11b4dbc24c085..6200167a98354 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -3665,20 +3665,6 @@ bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII,
   return hasAny64BitVGPROperands(OpDesc, MII, ST);
 }
 
-unsigned getLdsDwGranularity(const MCSubtargetInfo &ST) {
-  if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
-    return 64;
-  if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
-    return 128;
-  if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
-    return 256;
-  if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
-    return 320;
-  if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
-    return 512;
-  return 64; // In sync with getAddressableLocalMemorySize
-}
-
 bool isPackedSingleSGPRFP32Inst(unsigned Opc) {
   switch (Opc) {
   case AMDGPU::V_PK_ADD_F32_gfx1250:
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index e124073f0466e..a9358689514bc 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -1812,11 +1812,6 @@ getVGPRLoweringOperandTables(const MCInstrDesc &Desc);
 /// \returns true if a memory instruction supports scale_offset modifier.
 bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode);
 
-/// \returns lds block size in terms of dwords. \p
-/// This is used to calculate the lds size encoded for PAL metadata 3.0+ which
-/// must be defined in terms of bytes.
-unsigned getLdsDwGranularity(const MCSubtargetInfo &ST);
-
 class ClusterDimsAttr {
 public:
   enum class Kind { Unknown, NoCluster, VariableDims, FixedDims };
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index 6cd174d0723a7..e81d0d964aebe 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -556,6 +556,34 @@ unsigned AMDGPU::getLDSAllocGranule(Triple::SubArchType SubArch) {
   return getLDSAllocGranule(getGPUKindFromSubArch(SubArch));
 }
 
+unsigned AMDGPU::getLDSEncodingGranule(GPUKind AK) {
+  const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
+  if (Features.none())
+    return 256;
+  assert((Features.test(FEAT_LDS_ENCODING_GRANULARITY_256) ||
+          Features.test(FEAT_LDS_ENCODING_GRANULARITY_512) ||
+          Features.test(FEAT_LDS_ENCODING_GRANULARITY_1024) ||
+          Features.test(FEAT_LDS_ENCODING_GRANULARITY_1280) ||
+          Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048)) &&
+         "missing LDS encoding granularity feature");
+  if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_256))
+    return 256;
+  if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_512))
+    return 512;
+  if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_1024))
+    return 1024;
+  if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_1280))
+    return 1280;
+  if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048))
+    return 2048;
+
+  return 256;
+}
+
+unsigned AMDGPU::getLDSEncodingGranule(Triple::SubArchType SubArch) {
+  return getLDSEncodingGranule(getGPUKindFromSubArch(SubArch));
+}
+
 unsigned AMDGPU::getMaxWavesPerEU(GPUKind AK) {
   const GPUInfo *Info = getAMDGPUInfo(AK);
   return Info ? Info->MaxWavesPerEU : 10;
@@ -597,7 +625,12 @@ static const AMDGPUFeatureBitset FrontendOnlyFeatures = {
     FEAT_LDS_ALLOC_GRANULARITY_512,
     FEAT_LDS_ALLOC_GRANULARITY_1024,
     FEAT_LDS_ALLOC_GRANULARITY_1280,
-    FEAT_LDS_ALLOC_GRANULARITY_2048};
+    FEAT_LDS_ALLOC_GRANULARITY_2048,
+    FEAT_LDS_ENCODING_GRANULARITY_256,
+    FEAT_LDS_ENCODING_GRANULARITY_512,
+    FEAT_LDS_ENCODING_GRANULARITY_1024,
+    FEAT_LDS_ENCODING_GRANULARITY_1280,
+    FEAT_LDS_ENCODING_GRANULARITY_2048};
 
 // Add a GPU's features (minus the frontend-only ones) to \p Features. With \p
 // Overwrite false, existing entries are kept so user -mattr overrides win.
diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll b/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll
new file mode 100644
index 0000000000000..637add975dc9f
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll
@@ -0,0 +1,26 @@
+; RUN: llc -mtriple=amdgpu10.30-mesa-mesa3d < %s | FileCheck %s --check-prefixes=CHECK,MESA
+; RUN: llc -mtriple=amdgpu10.30-amd-amdpal < %s | FileCheck %s --check-prefixes=CHECK,PAL
+
+; gfx1030 allocates LDS in 1024-byte blocks but encodes it in 512-byte units.
+; 8252 bytes occupies 9216 bytes of LDS for occupancy, while its encoded size is
+; 17 units (8704 bytes). Using the allocation granule for either encoding step
+; would produce an incorrect block count or PAL metadata size.
+
+ at lds = addrspace(3) global [2063 x i32] poison, align 4
+
+; CHECK-LABEL: lds_granularity:
+; MESA: granulated_lds_size = 17
+; CHECK: ; Occupancy: 7{{$}}
+; PAL: .hardware_stages:
+; PAL: .cs:
+; PAL: .lds_size:       0x2200
+define amdgpu_kernel void @lds_granularity(i32 %index, i32 %value) #0 {
+  %ptr = getelementptr [2063 x i32], ptr addrspace(3) @lds, i32 0, i32 %index
+  store volatile i32 %value, ptr addrspace(3) %ptr
+  ret void
+}
+
+attributes #0 = { "amdgpu-flat-work-group-size"="1,64" }
+
+!amdgpu.pal.metadata.msgpack = !{!0}
+!0 = !{!"\81\AEamdpal.version\92\03\00"}
diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll b/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll
new file mode 100644
index 0000000000000..c02e454ee10b3
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll
@@ -0,0 +1,34 @@
+; RUN: llc -mtriple=amdgpu9.42-mesa-mesa3d < %s | FileCheck %s --check-prefix=GFX942
+; RUN: llc -mtriple=amdgpu9.50-mesa-mesa3d < %s | FileCheck %s --check-prefix=GFX950
+; RUN: llc -mtriple=amdgpu9.4-mesa-mesa3d < %s | FileCheck %s --check-prefix=GENERIC
+
+; gfx9-4-generic uses gfx950's 1280-byte allocation and encoding granules
+; independently of its 64 KiB addressable LDS capacity. gfx942 uses 512 bytes
+; for both granularities.
+
+ at lds1280 = addrspace(3) global [320 x i32] poison, align 4
+ at lds1284 = addrspace(3) global [321 x i32] poison, align 4
+
+; GFX942-LABEL: one_granule:
+; GFX942: granulated_lds_size = 3
+; GFX950-LABEL: one_granule:
+; GFX950: granulated_lds_size = 1
+; GENERIC-LABEL: one_granule:
+; GENERIC: granulated_lds_size = 1
+define amdgpu_kernel void @one_granule(i32 %index, i32 %value) {
+  %ptr = getelementptr [320 x i32], ptr addrspace(3) @lds1280, i32 0, i32 %index
+  store volatile i32 %value, ptr addrspace(3) %ptr
+  ret void
+}
+
+; GFX942-LABEL: two_granules:
+; GFX942: granulated_lds_size = 3
+; GFX950-LABEL: two_granules:
+; GFX950: granulated_lds_size = 2
+; GENERIC-LABEL: two_granules:
+; GENERIC: granulated_lds_size = 2
+define amdgpu_kernel void @two_granules(i32 %index, i32 %value) {
+  %ptr = getelementptr [321 x i32], ptr addrspace(3) @lds1284, i32 0, i32 %index
+  store volatile i32 %value, ptr addrspace(3) %ptr
+  ret void
+}
diff --git a/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td b/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
index 13136a7d00b59..b5e4337081a7d 100644
--- a/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
+++ b/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
@@ -3,6 +3,8 @@
 // RUN:   | FileCheck %t/valid.td
 // RUN: not llvm-tblgen -gen-amdgpu-target-def -I %t -I %p/../../include -I %p/../../lib/Target/AMDGPU %t/missing.td 2>&1 \
 // RUN:   | FileCheck %t/missing.td -DFILE=%t/missing.td --implicit-check-not="error:"
+// RUN: not llvm-tblgen -gen-amdgpu-target-def -I %t -I %p/../../include -I %p/../../lib/Target/AMDGPU %t/missing-encoding.td 2>&1 \
+// RUN:   | FileCheck %t/missing-encoding.td -DFILE=%t/missing-encoding.td --implicit-check-not="error:"
 
 //--- common.td
 include "llvm/Target/Target.td"
@@ -12,25 +14,42 @@ def MyTarget : Target;
 def AMDGPUFrontendVisibleFeatures {
   list<SubtargetFeature> Features = [FeatureLDSAllocGranularity512,
                                    FeatureLDSAllocGranularity1024,
-                                   FeatureLDSAllocGranularity1280];
+                                   FeatureLDSAllocGranularity1280,
+                                   FeatureLDSEncodingGranularity512,
+                                   FeatureLDSEncodingGranularity1024,
+                                   FeatureLDSEncodingGranularity1280];
 }
 
 def GFX942 : AMDGPUProcessorModel<"gfx942", NoSchedModel,
-    [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512],
+    [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512,
+     FeatureLDSEncodingGranularity512],
     [9, 4, 2]>;
 def GFX950 : AMDGPUProcessorModel<"gfx950", NoSchedModel,
-    [FeatureAddressableLocalMemorySize163840, FeatureLDSAllocGranularity1280],
+    [FeatureAddressableLocalMemorySize163840, FeatureLDSAllocGranularity1280,
+     FeatureLDSEncodingGranularity1280],
     [9, 5, 0]>;
 
 //--- valid.td
 include "common.td"
 // Capacity and granularity can each come from a different covered GPU.
 def : AMDGPUProcessorModel<"gfx9-4-generic", NoSchedModel,
-    [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1280],
+    [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1280,
+     FeatureLDSEncodingGranularity1280],
     [9, 4, 0]> {
   let CoveredGPUs = [GFX942, GFX950];
 }
-// CHECK: Triple::AMDGPUSubArch9_4, AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1280}), {9, 4, 0}, [[#]], 10, 65536, 32, 0},
+// CHECK: Triple::AMDGPUSubArch9_4, AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1280, FEAT_LDS_ENCODING_GRANULARITY_1280}), {9, 4, 0}, [[#]], 10, 65536, 32, 0},
+
+// Allocation and encoding use different units on gfx10.3, including its generic.
+def GFX1030 : AMDGPUProcessorModel<"gfx1030", NoSchedModel,
+    [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512],
+    [10, 3, 0]>;
+def : AMDGPUProcessorModel<"gfx10-3-generic", NoSchedModel,
+    [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512],
+    [10, 3, 0]> {
+  let CoveredGPUs = [GFX1030];
+}
+// CHECK: Triple::AMDGPUSubArch10_3, AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1024, FEAT_LDS_ENCODING_GRANULARITY_512})
 
 //--- missing.td
 include "common.td"
@@ -40,3 +59,16 @@ def : AMDGPUProcessorModel<"gfx9-4-generic", NoSchedModel,
     [9, 4, 0]> {
   let CoveredGPUs = [GFX942, GFX950];
 }
+
+//--- missing-encoding.td
+include "common.td"
+// An allocation feature does not provide support for an encoding feature.
+def GFX1030 : AMDGPUProcessorModel<"gfx1030", NoSchedModel,
+    [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512],
+    [10, 3, 0]>;
+// CHECK: [[FILE]]:[[#@LINE+1]]:1: error: generic target 'gfx10-3-generic' exposes feature 'lds-encoding-granularity-1024' not supported by any covered GPU
+def : AMDGPUProcessorModel<"gfx10-3-generic", NoSchedModel,
+    [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity1024],
+    [10, 3, 0]> {
+  let CoveredGPUs = [GFX1030];
+}
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 493e0a6215f4b..d381e8e949f74 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -2833,6 +2833,13 @@ TEST(TargetParserTest, testAMDGPUfillAMDGPUFeatureMap) {
   EXPECT_FALSE(HasFeature("gfx950", "lds-alloc-granularity-1280"));
   EXPECT_FALSE(HasFeature("gfx1310", "lds-alloc-granularity-1024"));
   EXPECT_FALSE(HasFeature("gfx1250", "lds-alloc-granularity-2048"));
+
+  // Encoding granularity is also queried through the bitset only.
+  EXPECT_FALSE(HasFeature("gfx600", "lds-encoding-granularity-256"));
+  EXPECT_FALSE(HasFeature("gfx1030", "lds-encoding-granularity-512"));
+  EXPECT_FALSE(HasFeature("gfx950", "lds-encoding-granularity-1280"));
+  EXPECT_FALSE(HasFeature("gfx1310", "lds-encoding-granularity-1024"));
+  EXPECT_FALSE(HasFeature("gfx1250", "lds-encoding-granularity-2048"));
 }
 
 TEST(TargetParserTest, testAMDGPUgetFeatureBitset) {
@@ -2887,30 +2894,41 @@ TEST(TargetParserTest, testAMDGPUHalfAddressableLDSFeature) {
   EXPECT_FALSE(Has(AMDGPU::GK_GFX1310));
 }
 
-TEST(TargetParserTest, testAMDGPULDSAllocGranularityFeatures) {
+TEST(TargetParserTest, testAMDGPULDSGranularityFeatures) {
   auto Has = [](AMDGPU::GPUKind AK, AMDGPU::AMDGPUFeature Feature) {
     return AMDGPU::getFeatureBitset(AK).test(Feature);
   };
-  auto Count = [&Has](AMDGPU::GPUKind AK) {
+  auto CountAlloc = [&Has](AMDGPU::GPUKind AK) {
     return Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_256) +
            Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512) +
            Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024) +
            Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1280) +
            Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_2048);
   };
+  auto CountEncoding = [&Has](AMDGPU::GPUKind AK) {
+    return Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_256) +
+           Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_512) +
+           Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1024) +
+           Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1280) +
+           Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_2048);
+  };
 
-  // Exactly one allocation granularity is set per GPU.
+  // Exactly one granularity of each kind is set per GPU, including generics.
   SmallVector<StringRef> AllGPUs;
   AMDGPU::fillValidArchListAMDGCN(AllGPUs, Triple::NoSubArch);
   for (StringRef Name : AllGPUs) {
     AMDGPU::GPUKind Kind = AMDGPU::parseArchAMDGCN(Name);
-    if (!AMDGPU::isPseudoTarget(Kind))
-      EXPECT_EQ(Count(Kind), 1) << Name;
+    if (!AMDGPU::isPseudoTarget(Kind)) {
+      EXPECT_EQ(CountAlloc(Kind), 1) << Name;
+      EXPECT_EQ(CountEncoding(Kind), 1) << Name;
+    }
   }
 
   // The legacy pseudo-targets do not represent hardware.
-  EXPECT_EQ(Count(AMDGPU::GK_GENERIC), 0);
-  EXPECT_EQ(Count(AMDGPU::GK_GENERIC_HSA), 0);
+  EXPECT_EQ(CountAlloc(AMDGPU::GK_GENERIC), 0);
+  EXPECT_EQ(CountAlloc(AMDGPU::GK_GENERIC_HSA), 0);
+  EXPECT_EQ(CountEncoding(AMDGPU::GK_GENERIC), 0);
+  EXPECT_EQ(CountEncoding(AMDGPU::GK_GENERIC_HSA), 0);
 
   EXPECT_TRUE(Has(AMDGPU::GK_GFX600, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_256));
   EXPECT_TRUE(Has(AMDGPU::GK_GFX900, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512));
@@ -2918,13 +2936,15 @@ TEST(TargetParserTest, testAMDGPULDSAllocGranularityFeatures) {
   EXPECT_TRUE(Has(AMDGPU::GK_GFX1310, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024));
   EXPECT_TRUE(Has(AMDGPU::GK_GFX1250, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_2048));
 
-  // RDNA2 and later 64 KiB targets allocate LDS in 1024-byte blocks.
+  // RDNA2 and later 64 KiB targets allocate 1024 bytes but encode 512-byte
+  // units.
   for (AMDGPU::GPUKind Kind :
        {AMDGPU::GK_GFX1030, AMDGPU::GK_GFX1100, AMDGPU::GK_GFX1170,
         AMDGPU::GK_GFX1200, AMDGPU::GK_GFX10_3_GENERIC,
         AMDGPU::GK_GFX11_GENERIC, AMDGPU::GK_GFX11_7_GENERIC,
         AMDGPU::GK_GFX12_GENERIC}) {
     EXPECT_TRUE(Has(Kind, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024));
+    EXPECT_TRUE(Has(Kind, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_512));
   }
 
   // A generic target uses the largest allocation granularity of the GPUs it
@@ -2933,6 +2953,8 @@ TEST(TargetParserTest, testAMDGPULDSAllocGranularityFeatures) {
       Has(AMDGPU::GK_GFX9_4_GENERIC, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1280));
   EXPECT_FALSE(
       Has(AMDGPU::GK_GFX9_4_GENERIC, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512));
+  EXPECT_TRUE(Has(AMDGPU::GK_GFX9_4_GENERIC,
+                  AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1280));
 }
 
 TEST(TargetParserTest, testAMDGPUfillValidArchListAMDGCN) {
@@ -3390,43 +3412,47 @@ TEST(TargetParserTest, testAMDGPUgetAddressableLocalMemorySize) {
       65536u);
 }
 
-TEST(TargetParserTest, testAMDGPUgetLDSAllocGranule) {
+TEST(TargetParserTest, testAMDGPUgetLDSGranules) {
   struct {
     AMDGPU::GPUKind Kind;
     Triple::SubArchType SubArch;
     unsigned Alloc;
+    unsigned Encoding;
   } Cases[] = {
-      {AMDGPU::GK_GFX600, Triple::AMDGPUSubArch600, 256},
-      {AMDGPU::GK_GFX700, Triple::AMDGPUSubArch700, 512},
-      {AMDGPU::GK_GFX900, Triple::AMDGPUSubArch900, 512},
-      {AMDGPU::GK_GFX942, Triple::AMDGPUSubArch942, 512},
-      {AMDGPU::GK_GFX950, Triple::AMDGPUSubArch950, 1280},
-      {AMDGPU::GK_GFX1010, Triple::AMDGPUSubArch1010, 512},
-      {AMDGPU::GK_GFX1030, Triple::AMDGPUSubArch1030, 1024},
-      {AMDGPU::GK_GFX1100, Triple::AMDGPUSubArch1100, 1024},
-      {AMDGPU::GK_GFX1150, Triple::AMDGPUSubArch1150, 1024},
-      {AMDGPU::GK_GFX1170, Triple::AMDGPUSubArch1170, 1024},
-      {AMDGPU::GK_GFX1200, Triple::AMDGPUSubArch1200, 1024},
-      {AMDGPU::GK_GFX1250, Triple::AMDGPUSubArch1250, 2048},
-      {AMDGPU::GK_GFX1251, Triple::AMDGPUSubArch1251, 2048},
-      {AMDGPU::GK_GFX1310, Triple::AMDGPUSubArch1310, 1024},
-      {AMDGPU::GK_GFX9_4_GENERIC, Triple::AMDGPUSubArch9_4, 1280},
-      {AMDGPU::GK_GFX10_1_GENERIC, Triple::AMDGPUSubArch10_1, 512},
-      {AMDGPU::GK_GFX10_3_GENERIC, Triple::AMDGPUSubArch10_3, 1024},
-      {AMDGPU::GK_GFX11_GENERIC, Triple::AMDGPUSubArch11, 1024},
-      {AMDGPU::GK_GFX11_7_GENERIC, Triple::AMDGPUSubArch11_7, 1024},
-      {AMDGPU::GK_GFX12_GENERIC, Triple::AMDGPUSubArch12, 1024},
-      {AMDGPU::GK_GFX12_5_GENERIC, Triple::AMDGPUSubArch12_5, 2048},
-      {AMDGPU::GK_NONE, Triple::NoSubArch, 256},
+      {AMDGPU::GK_GFX600, Triple::AMDGPUSubArch600, 256, 256},
+      {AMDGPU::GK_GFX700, Triple::AMDGPUSubArch700, 512, 512},
+      {AMDGPU::GK_GFX900, Triple::AMDGPUSubArch900, 512, 512},
+      {AMDGPU::GK_GFX942, Triple::AMDGPUSubArch942, 512, 512},
+      {AMDGPU::GK_GFX950, Triple::AMDGPUSubArch950, 1280, 1280},
+      {AMDGPU::GK_GFX1010, Triple::AMDGPUSubArch1010, 512, 512},
+      {AMDGPU::GK_GFX1030, Triple::AMDGPUSubArch1030, 1024, 512},
+      {AMDGPU::GK_GFX1100, Triple::AMDGPUSubArch1100, 1024, 512},
+      {AMDGPU::GK_GFX1150, Triple::AMDGPUSubArch1150, 1024, 512},
+      {AMDGPU::GK_GFX1170, Triple::AMDGPUSubArch1170, 1024, 512},
+      {AMDGPU::GK_GFX1200, Triple::AMDGPUSubArch1200, 1024, 512},
+      {AMDGPU::GK_GFX1250, Triple::AMDGPUSubArch1250, 2048, 2048},
+      {AMDGPU::GK_GFX1251, Triple::AMDGPUSubArch1251, 2048, 2048},
+      {AMDGPU::GK_GFX1310, Triple::AMDGPUSubArch1310, 1024, 1024},
+      {AMDGPU::GK_GFX9_4_GENERIC, Triple::AMDGPUSubArch9_4, 1280, 1280},
+      {AMDGPU::GK_GFX10_1_GENERIC, Triple::AMDGPUSubArch10_1, 512, 512},
+      {AMDGPU::GK_GFX10_3_GENERIC, Triple::AMDGPUSubArch10_3, 1024, 512},
+      {AMDGPU::GK_GFX11_GENERIC, Triple::AMDGPUSubArch11, 1024, 512},
+      {AMDGPU::GK_GFX11_7_GENERIC, Triple::AMDGPUSubArch11_7, 1024, 512},
+      {AMDGPU::GK_GFX12_GENERIC, Triple::AMDGPUSubArch12, 1024, 512},
+      {AMDGPU::GK_GFX12_5_GENERIC, Triple::AMDGPUSubArch12_5, 2048, 2048},
+      {AMDGPU::GK_NONE, Triple::NoSubArch, 256, 256},
   };
   for (const auto &Case : Cases) {
     SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Case.Kind));
     EXPECT_EQ(AMDGPU::getLDSAllocGranule(Case.Kind), Case.Alloc);
     EXPECT_EQ(AMDGPU::getLDSAllocGranule(Case.SubArch), Case.Alloc);
+    EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Case.Kind), Case.Encoding);
+    EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Case.SubArch), Case.Encoding);
   }
 
   for (AMDGPU::GPUKind Kind : {AMDGPU::GK_GENERIC, AMDGPU::GK_GENERIC_HSA}) {
     EXPECT_EQ(AMDGPU::getLDSAllocGranule(Kind), 256u);
+    EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Kind), 256u);
   }
 }
 

>From 147b3d9cf0f2fdbef8cc6ff0a5e09d8eea423cb7 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sun, 20 Sep 2026 11:17:26 -0400
Subject: [PATCH 2/3] [AMDGPU] Return zero LDS encoding granularity for dummy
 targets

Remove the redundant early return and assertion from getLDSEncodingGranule. Return zero when the target has no encoding granularity feature, and document and test the result for unknown and legacy generic targets.

Keep the existing 256-byte default in the assembly printer so compiling without a GPU still produces valid LDS sizes. Test default-target Mesa and PAL encodings and HSA metadata.

Change-Id: Ia0c69a0d7ce2858b31ee64100d37ae34bba6c42c
---
 .../llvm/TargetParser/AMDGPUTargetParser.h    |  1 +
 llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp   | 14 +++++++---
 llvm/lib/TargetParser/AMDGPUTargetParser.cpp  | 10 +------
 .../CodeGen/AMDGPU/lds-size-default-device.ll | 28 +++++++++++++++++++
 .../TargetParser/TargetParserTest.cpp         |  4 +--
 5 files changed, 42 insertions(+), 15 deletions(-)
 create mode 100644 llvm/test/CodeGen/AMDGPU/lds-size-default-device.ll

diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index 1239baa200a73..042542ced349d 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -247,6 +247,7 @@ LLVM_ABI unsigned getLDSAllocGranule(Triple::SubArchType SubArch);
 
 /// \returns LDS size encoding granularity in bytes, used for program resource
 /// registers and metadata. This can differ from the allocation granularity.
+/// Returns zero if the target has no LDS encoding granularity feature.
 LLVM_ABI unsigned getLDSEncodingGranule(GPUKind AK);
 LLVM_ABI unsigned getLDSEncodingGranule(Triple::SubArchType SubArch);
 
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index f25448a81f457..c394d6290bfa7 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1254,6 +1254,14 @@ static const MCExpr *computeAccumOffset(const MCExpr *NumVGPR, MCContext &Ctx) {
   return MCBinaryExpr::createSub(DivCeil, ConstOne, Ctx);
 }
 
+static unsigned getLDSEncodingGranule(const GCNSubtarget &ST) {
+  unsigned Granule =
+      AMDGPU::getLDSEncodingGranule(ST.getTargetID().getGPUKind());
+  // The legacy generic targets have no encoding granularity feature. Preserve
+  // the default used for code generation when no GPU is specified.
+  return Granule ? Granule : 256;
+}
+
 void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
                                         const MachineFunction &MF) {
   const GCNSubtarget &STM = MF.getSubtarget<GCNSubtarget>();
@@ -1440,8 +1448,7 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
 
   ProgInfo.LDSSize = MFI->getLDSSize();
 
-  unsigned LDSGranularityBytes =
-      AMDGPU::getLDSEncodingGranule(STM.getTargetID().getGPUKind());
+  unsigned LDSGranularityBytes = getLDSEncodingGranule(STM);
   ProgInfo.LDSBlocks =
       alignTo(ProgInfo.LDSSize, LDSGranularityBytes) / LDSGranularityBytes;
 
@@ -1680,8 +1687,7 @@ static void EmitPALMetadataCommon(AMDGPUPALMetadata *MD,
 
   MD->updateHwStageMaximum(
       CC, ".lds_size",
-      (unsigned)(CurrentProgramInfo.LdsSize *
-                 AMDGPU::getLDSEncodingGranule(ST.getTargetID().getGPUKind())));
+      (unsigned)(CurrentProgramInfo.LdsSize * getLDSEncodingGranule(ST)));
 }
 
 // This is the equivalent of EmitProgramInfoSI above, but for when the OS type
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index e81d0d964aebe..756d665353c45 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -558,14 +558,6 @@ unsigned AMDGPU::getLDSAllocGranule(Triple::SubArchType SubArch) {
 
 unsigned AMDGPU::getLDSEncodingGranule(GPUKind AK) {
   const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
-  if (Features.none())
-    return 256;
-  assert((Features.test(FEAT_LDS_ENCODING_GRANULARITY_256) ||
-          Features.test(FEAT_LDS_ENCODING_GRANULARITY_512) ||
-          Features.test(FEAT_LDS_ENCODING_GRANULARITY_1024) ||
-          Features.test(FEAT_LDS_ENCODING_GRANULARITY_1280) ||
-          Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048)) &&
-         "missing LDS encoding granularity feature");
   if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_256))
     return 256;
   if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_512))
@@ -577,7 +569,7 @@ unsigned AMDGPU::getLDSEncodingGranule(GPUKind AK) {
   if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048))
     return 2048;
 
-  return 256;
+  return 0;
 }
 
 unsigned AMDGPU::getLDSEncodingGranule(Triple::SubArchType SubArch) {
diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-default-device.ll b/llvm/test/CodeGen/AMDGPU/lds-size-default-device.ll
new file mode 100644
index 0000000000000..532fdf857c370
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/lds-size-default-device.ll
@@ -0,0 +1,28 @@
+; RUN: llc -mtriple=amdgcn-mesa-mesa3d < %s | FileCheck %s --check-prefixes=CHECK,MESA
+; RUN: llc -mtriple=amdgcn-amd-amdpal < %s | FileCheck %s --check-prefixes=CHECK,PAL
+; RUN: llc -mtriple=amdgcn-amd-amdhsa < %s | FileCheck %s --check-prefixes=CHECK,HSA
+
+; The legacy generic and generic-hsa targets have no LDS encoding granularity
+; feature. Code generation still uses its default 256-byte encoding granule,
+; rounding 260 bytes up to two units (512 bytes) for registers and PAL metadata.
+; HSA metadata records the unrounded size in bytes.
+
+ at lds = addrspace(3) global [65 x i32] poison, align 4
+
+; CHECK-LABEL: lds_granularity:
+; MESA: granulated_lds_size = 2
+; HSA: .amdhsa_group_segment_fixed_size 260
+; PAL: .hardware_stages:
+; PAL: .cs:
+; PAL: .lds_size:       0x200
+define amdgpu_kernel void @lds_granularity(i32 %index, i32 %value) {
+  %ptr = getelementptr [65 x i32], ptr addrspace(3) @lds, i32 0, i32 %index
+  store volatile i32 %value, ptr addrspace(3) %ptr
+  ret void
+}
+
+!llvm.module.flags = !{!0}
+!0 = !{i32 1, !"amdhsa_code_object_version", i32 400}
+
+!amdgpu.pal.metadata.msgpack = !{!1}
+!1 = !{!"\81\AEamdpal.version\92\03\00"}
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index d381e8e949f74..62d6f3a4c955b 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -3440,7 +3440,7 @@ TEST(TargetParserTest, testAMDGPUgetLDSGranules) {
       {AMDGPU::GK_GFX11_7_GENERIC, Triple::AMDGPUSubArch11_7, 1024, 512},
       {AMDGPU::GK_GFX12_GENERIC, Triple::AMDGPUSubArch12, 1024, 512},
       {AMDGPU::GK_GFX12_5_GENERIC, Triple::AMDGPUSubArch12_5, 2048, 2048},
-      {AMDGPU::GK_NONE, Triple::NoSubArch, 256, 256},
+      {AMDGPU::GK_NONE, Triple::NoSubArch, 256, 0},
   };
   for (const auto &Case : Cases) {
     SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Case.Kind));
@@ -3452,7 +3452,7 @@ TEST(TargetParserTest, testAMDGPUgetLDSGranules) {
 
   for (AMDGPU::GPUKind Kind : {AMDGPU::GK_GENERIC, AMDGPU::GK_GENERIC_HSA}) {
     EXPECT_EQ(AMDGPU::getLDSAllocGranule(Kind), 256u);
-    EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Kind), 256u);
+    EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Kind), 0u);
   }
 }
 

>From 84291b02031ce730c82613be488dda1833f97757 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Mon, 21 Sep 2026 14:57:09 -0400
Subject: [PATCH 3/3] [AMDGPU] Use new triples in default LDS encoding test

Change-Id: I54360f63a4b7c64f968c4e8f81c61542a2856f69
---
 llvm/test/CodeGen/AMDGPU/lds-size-default-device.ll | 6 +++---
 1 file changed, 3 insertions(+), 3 deletions(-)

diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-default-device.ll b/llvm/test/CodeGen/AMDGPU/lds-size-default-device.ll
index 532fdf857c370..df1ee9ff96749 100644
--- a/llvm/test/CodeGen/AMDGPU/lds-size-default-device.ll
+++ b/llvm/test/CodeGen/AMDGPU/lds-size-default-device.ll
@@ -1,6 +1,6 @@
-; RUN: llc -mtriple=amdgcn-mesa-mesa3d < %s | FileCheck %s --check-prefixes=CHECK,MESA
-; RUN: llc -mtriple=amdgcn-amd-amdpal < %s | FileCheck %s --check-prefixes=CHECK,PAL
-; RUN: llc -mtriple=amdgcn-amd-amdhsa < %s | FileCheck %s --check-prefixes=CHECK,HSA
+; RUN: llc -mtriple=amdgpu-mesa-mesa3d < %s | FileCheck %s --check-prefixes=CHECK,MESA
+; RUN: llc -mtriple=amdgpu-amd-amdpal < %s | FileCheck %s --check-prefixes=CHECK,PAL
+; RUN: llc -mtriple=amdgpu-amd-amdhsa < %s | FileCheck %s --check-prefixes=CHECK,HSA
 
 ; The legacy generic and generic-hsa targets have no LDS encoding granularity
 ; feature. Code generation still uses its default 256-byte encoding granule,



More information about the llvm-commits mailing list