[llvm-branch-commits] [llvm] [AMDGPU] Add LDS encoding granularity to TargetParser (PR #224852)
Chinmay Deshpande via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Sat Sep 19 11:35:59 PDT 2026
https://github.com/chinmaydd created https://github.com/llvm/llvm-project/pull/224852
Model LDS encoding granularity with dedicated features and expose the byte-valued `getLDSEncodingGranule` query for GPUKind and subarch.
Migrate program resource register and PAL metadata encoding to the new query and remove `getLdsDwGranularity` from `AMDGPUBaseInfo`
>From 47a076d2d3a0a93b14229299f4756e8dda633532 Mon Sep 17 00:00:00 2001
From: Chinmay Deshpande <chdeshpa at amd.com>
Date: Sat, 19 Sep 2026 14:15:35 -0400
Subject: [PATCH] [AMDGPU] Add LDS encoding granularity to TargetParser
Model LDS encoding granularity with dedicated features and expose the
byte-valued getLDSEncodingGranule query for GPUKind and subarch. Keep
encoding independent of the hardware allocation granularity used for
occupancy; GFX10.3, GFX11 and GFX12.0 encode in 512-byte units while
allocating 1024-byte blocks.
Migrate program resource register and PAL metadata encoding to the new
query and remove getLdsDwGranularity from AMDGPUBaseInfo. gfx9-4-generic
uses gfx950's 1280-byte encoding granule independently of LDS capacity.
Test encoding queries, feature membership, generic-target validation and
encoded LDS sizes, including the GFX10.3 allocation/encoding distinction.
Change-Id: I9d3c2c041605e9a45fa8fbda09fc3460a74953ea
---
.../llvm/TargetParser/AMDGPUTargetParser.h | 5 ++
llvm/lib/Target/AMDGPU/AMDGPU.td | 20 +++++
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 7 +-
llvm/lib/Target/AMDGPU/AMDGPUFeatures.td | 15 ++++
llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h | 1 +
llvm/lib/Target/AMDGPU/R600Processors.td | 2 +
.../Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 14 ---
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 5 --
llvm/lib/TargetParser/AMDGPUTargetParser.cpp | 35 +++++++-
llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll | 26 ++++++
.../CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll | 34 +++++++
.../AMDGPUTargetDefLDSAllocGranularity.td | 42 +++++++--
.../TargetParser/TargetParserTest.cpp | 88 ++++++++++++-------
13 files changed, 235 insertions(+), 59 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll
create mode 100644 llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll
diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index 49245a4322ab6..1239baa200a73 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -245,6 +245,11 @@ LLVM_ABI unsigned getLDSBankCount(Triple::SubArchType SubArch);
LLVM_ABI unsigned getLDSAllocGranule(GPUKind AK);
LLVM_ABI unsigned getLDSAllocGranule(Triple::SubArchType SubArch);
+/// \returns LDS size encoding granularity in bytes, used for program resource
+/// registers and metadata. This can differ from the allocation granularity.
+LLVM_ABI unsigned getLDSEncodingGranule(GPUKind AK);
+LLVM_ABI unsigned getLDSEncodingGranule(Triple::SubArchType SubArch);
+
/// \returns Number of SIMDs a work-group's waves run on. All four SIMDs of the
/// functional block in full-SIMD mode, half of them otherwise.
constexpr unsigned getNumWorkGroupSIMDs(bool FullSIMDMode) {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index 186191d76cb38..8ddbd8a44707a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -1654,6 +1654,7 @@ def FeatureSouthernIslands : GCNSubtargetFeatureGeneration<"SOUTHERN_ISLANDS",
[FeatureFP64, FeatureAddressableLocalMemorySize32768,
FeatureHalfAddressablePhysicalLocalMemory,
FeatureLDSAllocGranularity256,
+ FeatureLDSEncodingGranularity256,
FeatureMIMG_R128,
FeatureWavefrontSize64, FeatureSupportsWave64, FeatureSMemTimeInst,
FeatureMadMacF32Insts,
@@ -1674,6 +1675,7 @@ def FeatureSeaIslands : GCNSubtargetFeatureGeneration<"SEA_ISLANDS",
"sea-islands",
[FeatureFP64, FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity512,
+ FeatureLDSEncodingGranularity512,
FeatureMIMG_R128,
FeatureWavefrontSize64, FeatureSupportsWave64, FeatureFlatAddressSpace,
FeatureCIInsts, FeatureMovrel, FeatureTrigReducedRange,
@@ -1696,6 +1698,7 @@ def FeatureVolcanicIslands : GCNSubtargetFeatureGeneration<"VOLCANIC_ISLANDS",
"volcanic-islands",
[FeatureFP64, FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity512,
+ FeatureLDSEncodingGranularity512,
FeatureMIMG_R128,
FeatureWavefrontSize64, FeatureSupportsWave64, FeatureFlatAddressSpace,
FeatureGCN3Encoding, FeatureCIInsts, Feature16BitInsts,
@@ -1746,6 +1749,7 @@ def FeatureGFX9 : GCNSubtargetFeatureGeneration<"GFX9",
def FeatureGFX10 : GCNSubtargetFeatureGeneration<"GFX10",
"gfx10",
[FeatureFP64, FeatureAddressableLocalMemorySize65536,
+ FeatureLDSEncodingGranularity512,
FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128,
FeatureSupportsWave32, FeatureSupportsWave64, FeatureSupportsWGP,
FeatureFlatAddressSpace,
@@ -1781,6 +1785,7 @@ def FeatureGFX11 : GCNSubtargetFeatureGeneration<"GFX11",
"gfx11",
[FeatureFP64, FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity1024,
+ FeatureLDSEncodingGranularity512,
FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128,
FeatureSupportsWave32, FeatureSupportsWave64, FeatureSupportsWGP,
FeatureFlatAddressSpace, Feature16BitInsts,
@@ -1951,6 +1956,7 @@ def FeatureISAVersion9_0_Common : FeatureSet<
[FeatureGFX9,
FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity512,
+ FeatureLDSEncodingGranularity512,
FeatureLDSBankCount32,
FeatureImageInsts,
FeatureMadMacF32Insts]>;
@@ -2101,6 +2107,7 @@ def FeatureISAVersion9_5_Common : FeatureSet<
!listconcat(FeatureISAVersion9_4_Common.Features,
[FeatureAddressableLocalMemorySize163840,
FeatureLDSAllocGranularity1280,
+ FeatureLDSEncodingGranularity1280,
FeatureLDSBankCount64,
FeatureFP8Insts,
FeatureFP8ConversionInsts,
@@ -2124,6 +2131,7 @@ def FeatureISAVersion9_4_2 : FeatureSet<
[
FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity512,
+ FeatureLDSEncodingGranularity512,
FeatureLDSBankCount32,
FeatureFP8Insts,
FeatureFP8ConversionInsts,
@@ -2136,6 +2144,7 @@ def FeatureISAVersion9_4_Generic : FeatureSet<
!listconcat(FeatureISAVersion9_4_Common.Features,
[FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity1280,
+ FeatureLDSEncodingGranularity1280,
FeatureLDSBankCount32,
FeatureRequiresCOV6])>;
@@ -2348,6 +2357,7 @@ def FeatureISAVersion12 : FeatureSet<
FeatureBackOffBarrier,
FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity1024,
+ FeatureLDSEncodingGranularity512,
FeatureHalfAddressablePhysicalLocalMemory,
FeatureLDSBankCount32,
FeatureDLInsts,
@@ -2507,6 +2517,7 @@ def FeatureISAVersion12_50_STRICT : FeatureSet<
[FeatureGFX1250_STRICT,
FeatureAddressableLocalMemorySize327680,
FeatureLDSAllocGranularity2048,
+ FeatureLDSEncodingGranularity2048,
FeatureLDSBankCount64,
FeatureWMMAN16Insts,
FeatureVOP3PX2IncrementsVaVdstTwice,
@@ -2538,6 +2549,7 @@ def FeatureISAVersion12_50 : FeatureSet<
!listconcat(FeatureISAVersion12_50_Common.Features,
[FeatureAddressableLocalMemorySize327680,
FeatureLDSAllocGranularity2048,
+ FeatureLDSEncodingGranularity2048,
FeatureLDSBankCount64,
FeatureWMMAN16Insts,
FeatureWMMAF4Insts,
@@ -2570,6 +2582,7 @@ def FeatureISAVersion12_51 : FeatureSet<
!listconcat(FeatureISAVersion12_50_Common.Features,
[FeatureAddressableLocalMemorySize327680,
FeatureLDSAllocGranularity2048,
+ FeatureLDSEncodingGranularity2048,
FeatureLDSBankCount64,
FeatureWMMAN16Insts,
FeatureWMMAF4Insts,
@@ -2610,6 +2623,7 @@ def FeatureISAVersion12_5_Generic: FeatureSet<
!listconcat(FeatureISAVersion12_50_Common.Features,
[FeatureAddressableLocalMemorySize327680,
FeatureLDSAllocGranularity2048,
+ FeatureLDSEncodingGranularity2048,
FeatureLDSBankCount64,
FeatureBlock16ConversionScaleInsts,
FeatureSetregVGPRMSBFixup,
@@ -2629,6 +2643,7 @@ def FeatureISAVersion13 : FeatureSet<
FeatureWaveMatchInsts,
FeatureAddressableLocalMemorySize196608,
FeatureLDSAllocGranularity1024,
+ FeatureLDSEncodingGranularity1024,
Feature64BitLiterals,
FeatureLDSBankCount32,
FeatureDLInsts,
@@ -3323,6 +3338,11 @@ def AMDGPUFrontendVisibleFeatures {
FeatureLDSAllocGranularity1024,
FeatureLDSAllocGranularity1280,
FeatureLDSAllocGranularity2048,
+ FeatureLDSEncodingGranularity256,
+ FeatureLDSEncodingGranularity512,
+ FeatureLDSEncodingGranularity1024,
+ FeatureLDSEncodingGranularity1280,
+ FeatureLDSEncodingGranularity2048,
];
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 11c482c83f1e1..f25448a81f457 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1440,7 +1440,8 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
ProgInfo.LDSSize = MFI->getLDSSize();
- unsigned LDSGranularityBytes = getLdsDwGranularity(STM) * 4;
+ unsigned LDSGranularityBytes =
+ AMDGPU::getLDSEncodingGranule(STM.getTargetID().getGPUKind());
ProgInfo.LDSBlocks =
alignTo(ProgInfo.LDSSize, LDSGranularityBytes) / LDSGranularityBytes;
@@ -1679,8 +1680,8 @@ static void EmitPALMetadataCommon(AMDGPUPALMetadata *MD,
MD->updateHwStageMaximum(
CC, ".lds_size",
- (unsigned)(CurrentProgramInfo.LdsSize * getLdsDwGranularity(ST) *
- sizeof(uint32_t)));
+ (unsigned)(CurrentProgramInfo.LdsSize *
+ AMDGPU::getLDSEncodingGranule(ST.getTargetID().getGPUKind())));
}
// This is the equivalent of EmitProgramInfoSI above, but for when the OS type
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td b/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
index 49c90acabb74b..cd3ac06f60fd1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
@@ -60,6 +60,21 @@ def FeatureLDSAllocGranularity1024 : SubtargetFeatureLDSAllocGranularity<1024>;
def FeatureLDSAllocGranularity1280 : SubtargetFeatureLDSAllocGranularity<1280>;
def FeatureLDSAllocGranularity2048 : SubtargetFeatureLDSAllocGranularity<2048>;
+// Units used to encode LDS size in program resource registers and metadata.
+// These can differ from the hardware allocation granularity. A generic target's
+// encoding granularity must be present on at least one covered GPU.
+class SubtargetFeatureLDSEncodingGranularity <int Granularity> : AMDGPUGenericAnyFeature <
+ "lds-encoding-granularity-"#Granularity,
+ "LDSEncodingGranularity",
+ !cast<string>(Granularity),
+ "LDS encoding granularity in bytes.">;
+
+def FeatureLDSEncodingGranularity256 : SubtargetFeatureLDSEncodingGranularity<256>;
+def FeatureLDSEncodingGranularity512 : SubtargetFeatureLDSEncodingGranularity<512>;
+def FeatureLDSEncodingGranularity1024 : SubtargetFeatureLDSEncodingGranularity<1024>;
+def FeatureLDSEncodingGranularity1280 : SubtargetFeatureLDSEncodingGranularity<1280>;
+def FeatureLDSEncodingGranularity2048 : SubtargetFeatureLDSEncodingGranularity<2048>;
+
// Whether each wavefront size mode is available on the hardware,
// independent of the active mode.
def FeatureSupportsWave32 : SubtargetFeature<"supports-wave32",
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
index e1f331229c463..d2ab61671c932 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
@@ -60,6 +60,7 @@ class AMDGPUSubtarget {
unsigned LocalMemorySize = 0;
unsigned AddressableLocalMemorySize = 0;
unsigned LDSAllocationGranularity = 0;
+ unsigned LDSEncodingGranularity = 0;
char WavefrontSizeLog2 = 0;
unsigned FlatOffsetBitWidth = 0;
diff --git a/llvm/lib/Target/AMDGPU/R600Processors.td b/llvm/lib/Target/AMDGPU/R600Processors.td
index 04947b4c94b7c..a72c367fd74a2 100644
--- a/llvm/lib/Target/AMDGPU/R600Processors.td
+++ b/llvm/lib/Target/AMDGPU/R600Processors.td
@@ -61,6 +61,7 @@ def FeatureR700 : R600SubtargetFeatureGeneration<"R700", "r700",
def FeatureEvergreen : R600SubtargetFeatureGeneration<"EVERGREEN", "evergreen",
[FeatureFetchLimit16, FeatureAddressableLocalMemorySize32768,
FeatureLDSAllocGranularity256,
+ FeatureLDSEncodingGranularity256,
FeatureMadMacF32Insts]
>;
@@ -69,6 +70,7 @@ def FeatureNorthernIslands : R600SubtargetFeatureGeneration<"NORTHERN_ISLANDS",
[FeatureFetchLimit16, FeatureWavefrontSize64,
FeatureAddressableLocalMemorySize32768,
FeatureLDSAllocGranularity256,
+ FeatureLDSEncodingGranularity256,
FeatureMadMacF32Insts]
>;
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 11b4dbc24c085..6200167a98354 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -3665,20 +3665,6 @@ bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII,
return hasAny64BitVGPROperands(OpDesc, MII, ST);
}
-unsigned getLdsDwGranularity(const MCSubtargetInfo &ST) {
- if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
- return 64;
- if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
- return 128;
- if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
- return 256;
- if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
- return 320;
- if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
- return 512;
- return 64; // In sync with getAddressableLocalMemorySize
-}
-
bool isPackedSingleSGPRFP32Inst(unsigned Opc) {
switch (Opc) {
case AMDGPU::V_PK_ADD_F32_gfx1250:
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index e124073f0466e..a9358689514bc 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -1812,11 +1812,6 @@ getVGPRLoweringOperandTables(const MCInstrDesc &Desc);
/// \returns true if a memory instruction supports scale_offset modifier.
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode);
-/// \returns lds block size in terms of dwords. \p
-/// This is used to calculate the lds size encoded for PAL metadata 3.0+ which
-/// must be defined in terms of bytes.
-unsigned getLdsDwGranularity(const MCSubtargetInfo &ST);
-
class ClusterDimsAttr {
public:
enum class Kind { Unknown, NoCluster, VariableDims, FixedDims };
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index 6cd174d0723a7..e81d0d964aebe 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -556,6 +556,34 @@ unsigned AMDGPU::getLDSAllocGranule(Triple::SubArchType SubArch) {
return getLDSAllocGranule(getGPUKindFromSubArch(SubArch));
}
+unsigned AMDGPU::getLDSEncodingGranule(GPUKind AK) {
+ const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
+ if (Features.none())
+ return 256;
+ assert((Features.test(FEAT_LDS_ENCODING_GRANULARITY_256) ||
+ Features.test(FEAT_LDS_ENCODING_GRANULARITY_512) ||
+ Features.test(FEAT_LDS_ENCODING_GRANULARITY_1024) ||
+ Features.test(FEAT_LDS_ENCODING_GRANULARITY_1280) ||
+ Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048)) &&
+ "missing LDS encoding granularity feature");
+ if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_256))
+ return 256;
+ if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_512))
+ return 512;
+ if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_1024))
+ return 1024;
+ if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_1280))
+ return 1280;
+ if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048))
+ return 2048;
+
+ return 256;
+}
+
+unsigned AMDGPU::getLDSEncodingGranule(Triple::SubArchType SubArch) {
+ return getLDSEncodingGranule(getGPUKindFromSubArch(SubArch));
+}
+
unsigned AMDGPU::getMaxWavesPerEU(GPUKind AK) {
const GPUInfo *Info = getAMDGPUInfo(AK);
return Info ? Info->MaxWavesPerEU : 10;
@@ -597,7 +625,12 @@ static const AMDGPUFeatureBitset FrontendOnlyFeatures = {
FEAT_LDS_ALLOC_GRANULARITY_512,
FEAT_LDS_ALLOC_GRANULARITY_1024,
FEAT_LDS_ALLOC_GRANULARITY_1280,
- FEAT_LDS_ALLOC_GRANULARITY_2048};
+ FEAT_LDS_ALLOC_GRANULARITY_2048,
+ FEAT_LDS_ENCODING_GRANULARITY_256,
+ FEAT_LDS_ENCODING_GRANULARITY_512,
+ FEAT_LDS_ENCODING_GRANULARITY_1024,
+ FEAT_LDS_ENCODING_GRANULARITY_1280,
+ FEAT_LDS_ENCODING_GRANULARITY_2048};
// Add a GPU's features (minus the frontend-only ones) to \p Features. With \p
// Overwrite false, existing entries are kept so user -mattr overrides win.
diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll b/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll
new file mode 100644
index 0000000000000..637add975dc9f
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll
@@ -0,0 +1,26 @@
+; RUN: llc -mtriple=amdgpu10.30-mesa-mesa3d < %s | FileCheck %s --check-prefixes=CHECK,MESA
+; RUN: llc -mtriple=amdgpu10.30-amd-amdpal < %s | FileCheck %s --check-prefixes=CHECK,PAL
+
+; gfx1030 allocates LDS in 1024-byte blocks but encodes it in 512-byte units.
+; 8252 bytes occupies 9216 bytes of LDS for occupancy, while its encoded size is
+; 17 units (8704 bytes). Using the allocation granule for either encoding step
+; would produce an incorrect block count or PAL metadata size.
+
+ at lds = addrspace(3) global [2063 x i32] poison, align 4
+
+; CHECK-LABEL: lds_granularity:
+; MESA: granulated_lds_size = 17
+; CHECK: ; Occupancy: 7{{$}}
+; PAL: .hardware_stages:
+; PAL: .cs:
+; PAL: .lds_size: 0x2200
+define amdgpu_kernel void @lds_granularity(i32 %index, i32 %value) #0 {
+ %ptr = getelementptr [2063 x i32], ptr addrspace(3) @lds, i32 0, i32 %index
+ store volatile i32 %value, ptr addrspace(3) %ptr
+ ret void
+}
+
+attributes #0 = { "amdgpu-flat-work-group-size"="1,64" }
+
+!amdgpu.pal.metadata.msgpack = !{!0}
+!0 = !{!"\81\AEamdpal.version\92\03\00"}
diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll b/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll
new file mode 100644
index 0000000000000..c02e454ee10b3
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll
@@ -0,0 +1,34 @@
+; RUN: llc -mtriple=amdgpu9.42-mesa-mesa3d < %s | FileCheck %s --check-prefix=GFX942
+; RUN: llc -mtriple=amdgpu9.50-mesa-mesa3d < %s | FileCheck %s --check-prefix=GFX950
+; RUN: llc -mtriple=amdgpu9.4-mesa-mesa3d < %s | FileCheck %s --check-prefix=GENERIC
+
+; gfx9-4-generic uses gfx950's 1280-byte allocation and encoding granules
+; independently of its 64 KiB addressable LDS capacity. gfx942 uses 512 bytes
+; for both granularities.
+
+ at lds1280 = addrspace(3) global [320 x i32] poison, align 4
+ at lds1284 = addrspace(3) global [321 x i32] poison, align 4
+
+; GFX942-LABEL: one_granule:
+; GFX942: granulated_lds_size = 3
+; GFX950-LABEL: one_granule:
+; GFX950: granulated_lds_size = 1
+; GENERIC-LABEL: one_granule:
+; GENERIC: granulated_lds_size = 1
+define amdgpu_kernel void @one_granule(i32 %index, i32 %value) {
+ %ptr = getelementptr [320 x i32], ptr addrspace(3) @lds1280, i32 0, i32 %index
+ store volatile i32 %value, ptr addrspace(3) %ptr
+ ret void
+}
+
+; GFX942-LABEL: two_granules:
+; GFX942: granulated_lds_size = 3
+; GFX950-LABEL: two_granules:
+; GFX950: granulated_lds_size = 2
+; GENERIC-LABEL: two_granules:
+; GENERIC: granulated_lds_size = 2
+define amdgpu_kernel void @two_granules(i32 %index, i32 %value) {
+ %ptr = getelementptr [321 x i32], ptr addrspace(3) @lds1284, i32 0, i32 %index
+ store volatile i32 %value, ptr addrspace(3) %ptr
+ ret void
+}
diff --git a/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td b/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
index 13136a7d00b59..b5e4337081a7d 100644
--- a/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
+++ b/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
@@ -3,6 +3,8 @@
// RUN: | FileCheck %t/valid.td
// RUN: not llvm-tblgen -gen-amdgpu-target-def -I %t -I %p/../../include -I %p/../../lib/Target/AMDGPU %t/missing.td 2>&1 \
// RUN: | FileCheck %t/missing.td -DFILE=%t/missing.td --implicit-check-not="error:"
+// RUN: not llvm-tblgen -gen-amdgpu-target-def -I %t -I %p/../../include -I %p/../../lib/Target/AMDGPU %t/missing-encoding.td 2>&1 \
+// RUN: | FileCheck %t/missing-encoding.td -DFILE=%t/missing-encoding.td --implicit-check-not="error:"
//--- common.td
include "llvm/Target/Target.td"
@@ -12,25 +14,42 @@ def MyTarget : Target;
def AMDGPUFrontendVisibleFeatures {
list<SubtargetFeature> Features = [FeatureLDSAllocGranularity512,
FeatureLDSAllocGranularity1024,
- FeatureLDSAllocGranularity1280];
+ FeatureLDSAllocGranularity1280,
+ FeatureLDSEncodingGranularity512,
+ FeatureLDSEncodingGranularity1024,
+ FeatureLDSEncodingGranularity1280];
}
def GFX942 : AMDGPUProcessorModel<"gfx942", NoSchedModel,
- [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512],
+ [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512,
+ FeatureLDSEncodingGranularity512],
[9, 4, 2]>;
def GFX950 : AMDGPUProcessorModel<"gfx950", NoSchedModel,
- [FeatureAddressableLocalMemorySize163840, FeatureLDSAllocGranularity1280],
+ [FeatureAddressableLocalMemorySize163840, FeatureLDSAllocGranularity1280,
+ FeatureLDSEncodingGranularity1280],
[9, 5, 0]>;
//--- valid.td
include "common.td"
// Capacity and granularity can each come from a different covered GPU.
def : AMDGPUProcessorModel<"gfx9-4-generic", NoSchedModel,
- [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1280],
+ [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1280,
+ FeatureLDSEncodingGranularity1280],
[9, 4, 0]> {
let CoveredGPUs = [GFX942, GFX950];
}
-// CHECK: Triple::AMDGPUSubArch9_4, AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1280}), {9, 4, 0}, [[#]], 10, 65536, 32, 0},
+// CHECK: Triple::AMDGPUSubArch9_4, AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1280, FEAT_LDS_ENCODING_GRANULARITY_1280}), {9, 4, 0}, [[#]], 10, 65536, 32, 0},
+
+// Allocation and encoding use different units on gfx10.3, including its generic.
+def GFX1030 : AMDGPUProcessorModel<"gfx1030", NoSchedModel,
+ [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512],
+ [10, 3, 0]>;
+def : AMDGPUProcessorModel<"gfx10-3-generic", NoSchedModel,
+ [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512],
+ [10, 3, 0]> {
+ let CoveredGPUs = [GFX1030];
+}
+// CHECK: Triple::AMDGPUSubArch10_3, AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1024, FEAT_LDS_ENCODING_GRANULARITY_512})
//--- missing.td
include "common.td"
@@ -40,3 +59,16 @@ def : AMDGPUProcessorModel<"gfx9-4-generic", NoSchedModel,
[9, 4, 0]> {
let CoveredGPUs = [GFX942, GFX950];
}
+
+//--- missing-encoding.td
+include "common.td"
+// An allocation feature does not provide support for an encoding feature.
+def GFX1030 : AMDGPUProcessorModel<"gfx1030", NoSchedModel,
+ [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity512],
+ [10, 3, 0]>;
+// CHECK: [[FILE]]:[[#@LINE+1]]:1: error: generic target 'gfx10-3-generic' exposes feature 'lds-encoding-granularity-1024' not supported by any covered GPU
+def : AMDGPUProcessorModel<"gfx10-3-generic", NoSchedModel,
+ [FeatureLDSAllocGranularity1024, FeatureLDSEncodingGranularity1024],
+ [10, 3, 0]> {
+ let CoveredGPUs = [GFX1030];
+}
diff --git a/llvm/unittests/TargetParser/TargetParserTest.cpp b/llvm/unittests/TargetParser/TargetParserTest.cpp
index 493e0a6215f4b..d381e8e949f74 100644
--- a/llvm/unittests/TargetParser/TargetParserTest.cpp
+++ b/llvm/unittests/TargetParser/TargetParserTest.cpp
@@ -2833,6 +2833,13 @@ TEST(TargetParserTest, testAMDGPUfillAMDGPUFeatureMap) {
EXPECT_FALSE(HasFeature("gfx950", "lds-alloc-granularity-1280"));
EXPECT_FALSE(HasFeature("gfx1310", "lds-alloc-granularity-1024"));
EXPECT_FALSE(HasFeature("gfx1250", "lds-alloc-granularity-2048"));
+
+ // Encoding granularity is also queried through the bitset only.
+ EXPECT_FALSE(HasFeature("gfx600", "lds-encoding-granularity-256"));
+ EXPECT_FALSE(HasFeature("gfx1030", "lds-encoding-granularity-512"));
+ EXPECT_FALSE(HasFeature("gfx950", "lds-encoding-granularity-1280"));
+ EXPECT_FALSE(HasFeature("gfx1310", "lds-encoding-granularity-1024"));
+ EXPECT_FALSE(HasFeature("gfx1250", "lds-encoding-granularity-2048"));
}
TEST(TargetParserTest, testAMDGPUgetFeatureBitset) {
@@ -2887,30 +2894,41 @@ TEST(TargetParserTest, testAMDGPUHalfAddressableLDSFeature) {
EXPECT_FALSE(Has(AMDGPU::GK_GFX1310));
}
-TEST(TargetParserTest, testAMDGPULDSAllocGranularityFeatures) {
+TEST(TargetParserTest, testAMDGPULDSGranularityFeatures) {
auto Has = [](AMDGPU::GPUKind AK, AMDGPU::AMDGPUFeature Feature) {
return AMDGPU::getFeatureBitset(AK).test(Feature);
};
- auto Count = [&Has](AMDGPU::GPUKind AK) {
+ auto CountAlloc = [&Has](AMDGPU::GPUKind AK) {
return Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_256) +
Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512) +
Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024) +
Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1280) +
Has(AK, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_2048);
};
+ auto CountEncoding = [&Has](AMDGPU::GPUKind AK) {
+ return Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_256) +
+ Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_512) +
+ Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1024) +
+ Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1280) +
+ Has(AK, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_2048);
+ };
- // Exactly one allocation granularity is set per GPU.
+ // Exactly one granularity of each kind is set per GPU, including generics.
SmallVector<StringRef> AllGPUs;
AMDGPU::fillValidArchListAMDGCN(AllGPUs, Triple::NoSubArch);
for (StringRef Name : AllGPUs) {
AMDGPU::GPUKind Kind = AMDGPU::parseArchAMDGCN(Name);
- if (!AMDGPU::isPseudoTarget(Kind))
- EXPECT_EQ(Count(Kind), 1) << Name;
+ if (!AMDGPU::isPseudoTarget(Kind)) {
+ EXPECT_EQ(CountAlloc(Kind), 1) << Name;
+ EXPECT_EQ(CountEncoding(Kind), 1) << Name;
+ }
}
// The legacy pseudo-targets do not represent hardware.
- EXPECT_EQ(Count(AMDGPU::GK_GENERIC), 0);
- EXPECT_EQ(Count(AMDGPU::GK_GENERIC_HSA), 0);
+ EXPECT_EQ(CountAlloc(AMDGPU::GK_GENERIC), 0);
+ EXPECT_EQ(CountAlloc(AMDGPU::GK_GENERIC_HSA), 0);
+ EXPECT_EQ(CountEncoding(AMDGPU::GK_GENERIC), 0);
+ EXPECT_EQ(CountEncoding(AMDGPU::GK_GENERIC_HSA), 0);
EXPECT_TRUE(Has(AMDGPU::GK_GFX600, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_256));
EXPECT_TRUE(Has(AMDGPU::GK_GFX900, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512));
@@ -2918,13 +2936,15 @@ TEST(TargetParserTest, testAMDGPULDSAllocGranularityFeatures) {
EXPECT_TRUE(Has(AMDGPU::GK_GFX1310, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024));
EXPECT_TRUE(Has(AMDGPU::GK_GFX1250, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_2048));
- // RDNA2 and later 64 KiB targets allocate LDS in 1024-byte blocks.
+ // RDNA2 and later 64 KiB targets allocate 1024 bytes but encode 512-byte
+ // units.
for (AMDGPU::GPUKind Kind :
{AMDGPU::GK_GFX1030, AMDGPU::GK_GFX1100, AMDGPU::GK_GFX1170,
AMDGPU::GK_GFX1200, AMDGPU::GK_GFX10_3_GENERIC,
AMDGPU::GK_GFX11_GENERIC, AMDGPU::GK_GFX11_7_GENERIC,
AMDGPU::GK_GFX12_GENERIC}) {
EXPECT_TRUE(Has(Kind, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1024));
+ EXPECT_TRUE(Has(Kind, AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_512));
}
// A generic target uses the largest allocation granularity of the GPUs it
@@ -2933,6 +2953,8 @@ TEST(TargetParserTest, testAMDGPULDSAllocGranularityFeatures) {
Has(AMDGPU::GK_GFX9_4_GENERIC, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_1280));
EXPECT_FALSE(
Has(AMDGPU::GK_GFX9_4_GENERIC, AMDGPU::FEAT_LDS_ALLOC_GRANULARITY_512));
+ EXPECT_TRUE(Has(AMDGPU::GK_GFX9_4_GENERIC,
+ AMDGPU::FEAT_LDS_ENCODING_GRANULARITY_1280));
}
TEST(TargetParserTest, testAMDGPUfillValidArchListAMDGCN) {
@@ -3390,43 +3412,47 @@ TEST(TargetParserTest, testAMDGPUgetAddressableLocalMemorySize) {
65536u);
}
-TEST(TargetParserTest, testAMDGPUgetLDSAllocGranule) {
+TEST(TargetParserTest, testAMDGPUgetLDSGranules) {
struct {
AMDGPU::GPUKind Kind;
Triple::SubArchType SubArch;
unsigned Alloc;
+ unsigned Encoding;
} Cases[] = {
- {AMDGPU::GK_GFX600, Triple::AMDGPUSubArch600, 256},
- {AMDGPU::GK_GFX700, Triple::AMDGPUSubArch700, 512},
- {AMDGPU::GK_GFX900, Triple::AMDGPUSubArch900, 512},
- {AMDGPU::GK_GFX942, Triple::AMDGPUSubArch942, 512},
- {AMDGPU::GK_GFX950, Triple::AMDGPUSubArch950, 1280},
- {AMDGPU::GK_GFX1010, Triple::AMDGPUSubArch1010, 512},
- {AMDGPU::GK_GFX1030, Triple::AMDGPUSubArch1030, 1024},
- {AMDGPU::GK_GFX1100, Triple::AMDGPUSubArch1100, 1024},
- {AMDGPU::GK_GFX1150, Triple::AMDGPUSubArch1150, 1024},
- {AMDGPU::GK_GFX1170, Triple::AMDGPUSubArch1170, 1024},
- {AMDGPU::GK_GFX1200, Triple::AMDGPUSubArch1200, 1024},
- {AMDGPU::GK_GFX1250, Triple::AMDGPUSubArch1250, 2048},
- {AMDGPU::GK_GFX1251, Triple::AMDGPUSubArch1251, 2048},
- {AMDGPU::GK_GFX1310, Triple::AMDGPUSubArch1310, 1024},
- {AMDGPU::GK_GFX9_4_GENERIC, Triple::AMDGPUSubArch9_4, 1280},
- {AMDGPU::GK_GFX10_1_GENERIC, Triple::AMDGPUSubArch10_1, 512},
- {AMDGPU::GK_GFX10_3_GENERIC, Triple::AMDGPUSubArch10_3, 1024},
- {AMDGPU::GK_GFX11_GENERIC, Triple::AMDGPUSubArch11, 1024},
- {AMDGPU::GK_GFX11_7_GENERIC, Triple::AMDGPUSubArch11_7, 1024},
- {AMDGPU::GK_GFX12_GENERIC, Triple::AMDGPUSubArch12, 1024},
- {AMDGPU::GK_GFX12_5_GENERIC, Triple::AMDGPUSubArch12_5, 2048},
- {AMDGPU::GK_NONE, Triple::NoSubArch, 256},
+ {AMDGPU::GK_GFX600, Triple::AMDGPUSubArch600, 256, 256},
+ {AMDGPU::GK_GFX700, Triple::AMDGPUSubArch700, 512, 512},
+ {AMDGPU::GK_GFX900, Triple::AMDGPUSubArch900, 512, 512},
+ {AMDGPU::GK_GFX942, Triple::AMDGPUSubArch942, 512, 512},
+ {AMDGPU::GK_GFX950, Triple::AMDGPUSubArch950, 1280, 1280},
+ {AMDGPU::GK_GFX1010, Triple::AMDGPUSubArch1010, 512, 512},
+ {AMDGPU::GK_GFX1030, Triple::AMDGPUSubArch1030, 1024, 512},
+ {AMDGPU::GK_GFX1100, Triple::AMDGPUSubArch1100, 1024, 512},
+ {AMDGPU::GK_GFX1150, Triple::AMDGPUSubArch1150, 1024, 512},
+ {AMDGPU::GK_GFX1170, Triple::AMDGPUSubArch1170, 1024, 512},
+ {AMDGPU::GK_GFX1200, Triple::AMDGPUSubArch1200, 1024, 512},
+ {AMDGPU::GK_GFX1250, Triple::AMDGPUSubArch1250, 2048, 2048},
+ {AMDGPU::GK_GFX1251, Triple::AMDGPUSubArch1251, 2048, 2048},
+ {AMDGPU::GK_GFX1310, Triple::AMDGPUSubArch1310, 1024, 1024},
+ {AMDGPU::GK_GFX9_4_GENERIC, Triple::AMDGPUSubArch9_4, 1280, 1280},
+ {AMDGPU::GK_GFX10_1_GENERIC, Triple::AMDGPUSubArch10_1, 512, 512},
+ {AMDGPU::GK_GFX10_3_GENERIC, Triple::AMDGPUSubArch10_3, 1024, 512},
+ {AMDGPU::GK_GFX11_GENERIC, Triple::AMDGPUSubArch11, 1024, 512},
+ {AMDGPU::GK_GFX11_7_GENERIC, Triple::AMDGPUSubArch11_7, 1024, 512},
+ {AMDGPU::GK_GFX12_GENERIC, Triple::AMDGPUSubArch12, 1024, 512},
+ {AMDGPU::GK_GFX12_5_GENERIC, Triple::AMDGPUSubArch12_5, 2048, 2048},
+ {AMDGPU::GK_NONE, Triple::NoSubArch, 256, 256},
};
for (const auto &Case : Cases) {
SCOPED_TRACE(AMDGPU::getArchNameAMDGCN(Case.Kind));
EXPECT_EQ(AMDGPU::getLDSAllocGranule(Case.Kind), Case.Alloc);
EXPECT_EQ(AMDGPU::getLDSAllocGranule(Case.SubArch), Case.Alloc);
+ EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Case.Kind), Case.Encoding);
+ EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Case.SubArch), Case.Encoding);
}
for (AMDGPU::GPUKind Kind : {AMDGPU::GK_GENERIC, AMDGPU::GK_GENERIC_HSA}) {
EXPECT_EQ(AMDGPU::getLDSAllocGranule(Kind), 256u);
+ EXPECT_EQ(AMDGPU::getLDSEncodingGranule(Kind), 256u);
}
}
More information about the llvm-branch-commits
mailing list