[llvm-branch-commits] [llvm] [AMDGPU] Add LDS encoding granularity to TargetParser (PR #224852)
via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Sat Sep 19 11:36:42 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Chinmay Deshpande (chinmaydd)
<details>
<summary>Changes</summary>
Model LDS encoding granularity with dedicated features and expose the byte-valued `getLDSEncodingGranule` query for GPUKind and subarch.
Migrate program resource register and PAL metadata encoding to the new query and remove `getLdsDwGranularity` from `AMDGPUBaseInfo`
---
Patch is 29.61 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/224852.diff
13 Files Affected:
- (modified) llvm/include/llvm/TargetParser/AMDGPUTargetParser.h (+5)
- (modified) llvm/lib/Target/AMDGPU/AMDGPU.td (+20)
- (modified) llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp (+4-3)
- (modified) llvm/lib/Target/AMDGPU/AMDGPUFeatures.td (+15)
- (modified) llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h (+1)
- (modified) llvm/lib/Target/AMDGPU/R600Processors.td (+2)
- (modified) llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp (-14)
- (modified) llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h (-5)
- (modified) llvm/lib/TargetParser/AMDGPUTargetParser.cpp (+34-1)
- (added) llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll (+26)
- (added) llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll (+34)
- (modified) llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td (+37-5)
- (modified) llvm/unittests/TargetParser/TargetParserTest.cpp (+57-31)
``````````diff
diff --git a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
index 49245a4322ab6..1239baa200a73 100644
--- a/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
+++ b/llvm/include/llvm/TargetParser/AMDGPUTargetParser.h
@@ -245,6 +245,11 @@ LLVM_ABI unsigned getLDSBankCount(Triple::SubArchType SubArch);
LLVM_ABI unsigned getLDSAllocGranule(GPUKind AK);
LLVM_ABI unsigned getLDSAllocGranule(Triple::SubArchType SubArch);
+/// \returns LDS size encoding granularity in bytes, used for program resource
+/// registers and metadata. This can differ from the allocation granularity.
+LLVM_ABI unsigned getLDSEncodingGranule(GPUKind AK);
+LLVM_ABI unsigned getLDSEncodingGranule(Triple::SubArchType SubArch);
+
/// \returns Number of SIMDs a work-group's waves run on. All four SIMDs of the
/// functional block in full-SIMD mode, half of them otherwise.
constexpr unsigned getNumWorkGroupSIMDs(bool FullSIMDMode) {
diff --git a/llvm/lib/Target/AMDGPU/AMDGPU.td b/llvm/lib/Target/AMDGPU/AMDGPU.td
index 186191d76cb38..8ddbd8a44707a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPU.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPU.td
@@ -1654,6 +1654,7 @@ def FeatureSouthernIslands : GCNSubtargetFeatureGeneration<"SOUTHERN_ISLANDS",
[FeatureFP64, FeatureAddressableLocalMemorySize32768,
FeatureHalfAddressablePhysicalLocalMemory,
FeatureLDSAllocGranularity256,
+ FeatureLDSEncodingGranularity256,
FeatureMIMG_R128,
FeatureWavefrontSize64, FeatureSupportsWave64, FeatureSMemTimeInst,
FeatureMadMacF32Insts,
@@ -1674,6 +1675,7 @@ def FeatureSeaIslands : GCNSubtargetFeatureGeneration<"SEA_ISLANDS",
"sea-islands",
[FeatureFP64, FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity512,
+ FeatureLDSEncodingGranularity512,
FeatureMIMG_R128,
FeatureWavefrontSize64, FeatureSupportsWave64, FeatureFlatAddressSpace,
FeatureCIInsts, FeatureMovrel, FeatureTrigReducedRange,
@@ -1696,6 +1698,7 @@ def FeatureVolcanicIslands : GCNSubtargetFeatureGeneration<"VOLCANIC_ISLANDS",
"volcanic-islands",
[FeatureFP64, FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity512,
+ FeatureLDSEncodingGranularity512,
FeatureMIMG_R128,
FeatureWavefrontSize64, FeatureSupportsWave64, FeatureFlatAddressSpace,
FeatureGCN3Encoding, FeatureCIInsts, Feature16BitInsts,
@@ -1746,6 +1749,7 @@ def FeatureGFX9 : GCNSubtargetFeatureGeneration<"GFX9",
def FeatureGFX10 : GCNSubtargetFeatureGeneration<"GFX10",
"gfx10",
[FeatureFP64, FeatureAddressableLocalMemorySize65536,
+ FeatureLDSEncodingGranularity512,
FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128,
FeatureSupportsWave32, FeatureSupportsWave64, FeatureSupportsWGP,
FeatureFlatAddressSpace,
@@ -1781,6 +1785,7 @@ def FeatureGFX11 : GCNSubtargetFeatureGeneration<"GFX11",
"gfx11",
[FeatureFP64, FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity1024,
+ FeatureLDSEncodingGranularity512,
FeatureHalfAddressablePhysicalLocalMemory, FeatureMIMG_R128,
FeatureSupportsWave32, FeatureSupportsWave64, FeatureSupportsWGP,
FeatureFlatAddressSpace, Feature16BitInsts,
@@ -1951,6 +1956,7 @@ def FeatureISAVersion9_0_Common : FeatureSet<
[FeatureGFX9,
FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity512,
+ FeatureLDSEncodingGranularity512,
FeatureLDSBankCount32,
FeatureImageInsts,
FeatureMadMacF32Insts]>;
@@ -2101,6 +2107,7 @@ def FeatureISAVersion9_5_Common : FeatureSet<
!listconcat(FeatureISAVersion9_4_Common.Features,
[FeatureAddressableLocalMemorySize163840,
FeatureLDSAllocGranularity1280,
+ FeatureLDSEncodingGranularity1280,
FeatureLDSBankCount64,
FeatureFP8Insts,
FeatureFP8ConversionInsts,
@@ -2124,6 +2131,7 @@ def FeatureISAVersion9_4_2 : FeatureSet<
[
FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity512,
+ FeatureLDSEncodingGranularity512,
FeatureLDSBankCount32,
FeatureFP8Insts,
FeatureFP8ConversionInsts,
@@ -2136,6 +2144,7 @@ def FeatureISAVersion9_4_Generic : FeatureSet<
!listconcat(FeatureISAVersion9_4_Common.Features,
[FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity1280,
+ FeatureLDSEncodingGranularity1280,
FeatureLDSBankCount32,
FeatureRequiresCOV6])>;
@@ -2348,6 +2357,7 @@ def FeatureISAVersion12 : FeatureSet<
FeatureBackOffBarrier,
FeatureAddressableLocalMemorySize65536,
FeatureLDSAllocGranularity1024,
+ FeatureLDSEncodingGranularity512,
FeatureHalfAddressablePhysicalLocalMemory,
FeatureLDSBankCount32,
FeatureDLInsts,
@@ -2507,6 +2517,7 @@ def FeatureISAVersion12_50_STRICT : FeatureSet<
[FeatureGFX1250_STRICT,
FeatureAddressableLocalMemorySize327680,
FeatureLDSAllocGranularity2048,
+ FeatureLDSEncodingGranularity2048,
FeatureLDSBankCount64,
FeatureWMMAN16Insts,
FeatureVOP3PX2IncrementsVaVdstTwice,
@@ -2538,6 +2549,7 @@ def FeatureISAVersion12_50 : FeatureSet<
!listconcat(FeatureISAVersion12_50_Common.Features,
[FeatureAddressableLocalMemorySize327680,
FeatureLDSAllocGranularity2048,
+ FeatureLDSEncodingGranularity2048,
FeatureLDSBankCount64,
FeatureWMMAN16Insts,
FeatureWMMAF4Insts,
@@ -2570,6 +2582,7 @@ def FeatureISAVersion12_51 : FeatureSet<
!listconcat(FeatureISAVersion12_50_Common.Features,
[FeatureAddressableLocalMemorySize327680,
FeatureLDSAllocGranularity2048,
+ FeatureLDSEncodingGranularity2048,
FeatureLDSBankCount64,
FeatureWMMAN16Insts,
FeatureWMMAF4Insts,
@@ -2610,6 +2623,7 @@ def FeatureISAVersion12_5_Generic: FeatureSet<
!listconcat(FeatureISAVersion12_50_Common.Features,
[FeatureAddressableLocalMemorySize327680,
FeatureLDSAllocGranularity2048,
+ FeatureLDSEncodingGranularity2048,
FeatureLDSBankCount64,
FeatureBlock16ConversionScaleInsts,
FeatureSetregVGPRMSBFixup,
@@ -2629,6 +2643,7 @@ def FeatureISAVersion13 : FeatureSet<
FeatureWaveMatchInsts,
FeatureAddressableLocalMemorySize196608,
FeatureLDSAllocGranularity1024,
+ FeatureLDSEncodingGranularity1024,
Feature64BitLiterals,
FeatureLDSBankCount32,
FeatureDLInsts,
@@ -3323,6 +3338,11 @@ def AMDGPUFrontendVisibleFeatures {
FeatureLDSAllocGranularity1024,
FeatureLDSAllocGranularity1280,
FeatureLDSAllocGranularity2048,
+ FeatureLDSEncodingGranularity256,
+ FeatureLDSEncodingGranularity512,
+ FeatureLDSEncodingGranularity1024,
+ FeatureLDSEncodingGranularity1280,
+ FeatureLDSEncodingGranularity2048,
];
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index 11c482c83f1e1..f25448a81f457 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -1440,7 +1440,8 @@ void AMDGPUAsmPrinter::getSIProgramInfo(SIProgramInfo &ProgInfo,
ProgInfo.LDSSize = MFI->getLDSSize();
- unsigned LDSGranularityBytes = getLdsDwGranularity(STM) * 4;
+ unsigned LDSGranularityBytes =
+ AMDGPU::getLDSEncodingGranule(STM.getTargetID().getGPUKind());
ProgInfo.LDSBlocks =
alignTo(ProgInfo.LDSSize, LDSGranularityBytes) / LDSGranularityBytes;
@@ -1679,8 +1680,8 @@ static void EmitPALMetadataCommon(AMDGPUPALMetadata *MD,
MD->updateHwStageMaximum(
CC, ".lds_size",
- (unsigned)(CurrentProgramInfo.LdsSize * getLdsDwGranularity(ST) *
- sizeof(uint32_t)));
+ (unsigned)(CurrentProgramInfo.LdsSize *
+ AMDGPU::getLDSEncodingGranule(ST.getTargetID().getGPUKind())));
}
// This is the equivalent of EmitProgramInfoSI above, but for when the OS type
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td b/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
index 49c90acabb74b..cd3ac06f60fd1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
+++ b/llvm/lib/Target/AMDGPU/AMDGPUFeatures.td
@@ -60,6 +60,21 @@ def FeatureLDSAllocGranularity1024 : SubtargetFeatureLDSAllocGranularity<1024>;
def FeatureLDSAllocGranularity1280 : SubtargetFeatureLDSAllocGranularity<1280>;
def FeatureLDSAllocGranularity2048 : SubtargetFeatureLDSAllocGranularity<2048>;
+// Units used to encode LDS size in program resource registers and metadata.
+// These can differ from the hardware allocation granularity. A generic target's
+// encoding granularity must be present on at least one covered GPU.
+class SubtargetFeatureLDSEncodingGranularity <int Granularity> : AMDGPUGenericAnyFeature <
+ "lds-encoding-granularity-"#Granularity,
+ "LDSEncodingGranularity",
+ !cast<string>(Granularity),
+ "LDS encoding granularity in bytes.">;
+
+def FeatureLDSEncodingGranularity256 : SubtargetFeatureLDSEncodingGranularity<256>;
+def FeatureLDSEncodingGranularity512 : SubtargetFeatureLDSEncodingGranularity<512>;
+def FeatureLDSEncodingGranularity1024 : SubtargetFeatureLDSEncodingGranularity<1024>;
+def FeatureLDSEncodingGranularity1280 : SubtargetFeatureLDSEncodingGranularity<1280>;
+def FeatureLDSEncodingGranularity2048 : SubtargetFeatureLDSEncodingGranularity<2048>;
+
// Whether each wavefront size mode is available on the hardware,
// independent of the active mode.
def FeatureSupportsWave32 : SubtargetFeature<"supports-wave32",
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
index e1f331229c463..d2ab61671c932 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
@@ -60,6 +60,7 @@ class AMDGPUSubtarget {
unsigned LocalMemorySize = 0;
unsigned AddressableLocalMemorySize = 0;
unsigned LDSAllocationGranularity = 0;
+ unsigned LDSEncodingGranularity = 0;
char WavefrontSizeLog2 = 0;
unsigned FlatOffsetBitWidth = 0;
diff --git a/llvm/lib/Target/AMDGPU/R600Processors.td b/llvm/lib/Target/AMDGPU/R600Processors.td
index 04947b4c94b7c..a72c367fd74a2 100644
--- a/llvm/lib/Target/AMDGPU/R600Processors.td
+++ b/llvm/lib/Target/AMDGPU/R600Processors.td
@@ -61,6 +61,7 @@ def FeatureR700 : R600SubtargetFeatureGeneration<"R700", "r700",
def FeatureEvergreen : R600SubtargetFeatureGeneration<"EVERGREEN", "evergreen",
[FeatureFetchLimit16, FeatureAddressableLocalMemorySize32768,
FeatureLDSAllocGranularity256,
+ FeatureLDSEncodingGranularity256,
FeatureMadMacF32Insts]
>;
@@ -69,6 +70,7 @@ def FeatureNorthernIslands : R600SubtargetFeatureGeneration<"NORTHERN_ISLANDS",
[FeatureFetchLimit16, FeatureWavefrontSize64,
FeatureAddressableLocalMemorySize32768,
FeatureLDSAllocGranularity256,
+ FeatureLDSEncodingGranularity256,
FeatureMadMacF32Insts]
>;
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 11b4dbc24c085..6200167a98354 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -3665,20 +3665,6 @@ bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII,
return hasAny64BitVGPROperands(OpDesc, MII, ST);
}
-unsigned getLdsDwGranularity(const MCSubtargetInfo &ST) {
- if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
- return 64;
- if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
- return 128;
- if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
- return 256;
- if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
- return 320;
- if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
- return 512;
- return 64; // In sync with getAddressableLocalMemorySize
-}
-
bool isPackedSingleSGPRFP32Inst(unsigned Opc) {
switch (Opc) {
case AMDGPU::V_PK_ADD_F32_gfx1250:
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index e124073f0466e..a9358689514bc 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -1812,11 +1812,6 @@ getVGPRLoweringOperandTables(const MCInstrDesc &Desc);
/// \returns true if a memory instruction supports scale_offset modifier.
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode);
-/// \returns lds block size in terms of dwords. \p
-/// This is used to calculate the lds size encoded for PAL metadata 3.0+ which
-/// must be defined in terms of bytes.
-unsigned getLdsDwGranularity(const MCSubtargetInfo &ST);
-
class ClusterDimsAttr {
public:
enum class Kind { Unknown, NoCluster, VariableDims, FixedDims };
diff --git a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
index 6cd174d0723a7..e81d0d964aebe 100644
--- a/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
+++ b/llvm/lib/TargetParser/AMDGPUTargetParser.cpp
@@ -556,6 +556,34 @@ unsigned AMDGPU::getLDSAllocGranule(Triple::SubArchType SubArch) {
return getLDSAllocGranule(getGPUKindFromSubArch(SubArch));
}
+unsigned AMDGPU::getLDSEncodingGranule(GPUKind AK) {
+ const AMDGPUFeatureBitset &Features = getFeatureBitset(AK);
+ if (Features.none())
+ return 256;
+ assert((Features.test(FEAT_LDS_ENCODING_GRANULARITY_256) ||
+ Features.test(FEAT_LDS_ENCODING_GRANULARITY_512) ||
+ Features.test(FEAT_LDS_ENCODING_GRANULARITY_1024) ||
+ Features.test(FEAT_LDS_ENCODING_GRANULARITY_1280) ||
+ Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048)) &&
+ "missing LDS encoding granularity feature");
+ if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_256))
+ return 256;
+ if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_512))
+ return 512;
+ if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_1024))
+ return 1024;
+ if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_1280))
+ return 1280;
+ if (Features.test(FEAT_LDS_ENCODING_GRANULARITY_2048))
+ return 2048;
+
+ return 256;
+}
+
+unsigned AMDGPU::getLDSEncodingGranule(Triple::SubArchType SubArch) {
+ return getLDSEncodingGranule(getGPUKindFromSubArch(SubArch));
+}
+
unsigned AMDGPU::getMaxWavesPerEU(GPUKind AK) {
const GPUInfo *Info = getAMDGPUInfo(AK);
return Info ? Info->MaxWavesPerEU : 10;
@@ -597,7 +625,12 @@ static const AMDGPUFeatureBitset FrontendOnlyFeatures = {
FEAT_LDS_ALLOC_GRANULARITY_512,
FEAT_LDS_ALLOC_GRANULARITY_1024,
FEAT_LDS_ALLOC_GRANULARITY_1280,
- FEAT_LDS_ALLOC_GRANULARITY_2048};
+ FEAT_LDS_ALLOC_GRANULARITY_2048,
+ FEAT_LDS_ENCODING_GRANULARITY_256,
+ FEAT_LDS_ENCODING_GRANULARITY_512,
+ FEAT_LDS_ENCODING_GRANULARITY_1024,
+ FEAT_LDS_ENCODING_GRANULARITY_1280,
+ FEAT_LDS_ENCODING_GRANULARITY_2048};
// Add a GPU's features (minus the frontend-only ones) to \p Features. With \p
// Overwrite false, existing entries are kept so user -mattr overrides win.
diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll b/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll
new file mode 100644
index 0000000000000..637add975dc9f
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/lds-size-gfx1030.ll
@@ -0,0 +1,26 @@
+; RUN: llc -mtriple=amdgpu10.30-mesa-mesa3d < %s | FileCheck %s --check-prefixes=CHECK,MESA
+; RUN: llc -mtriple=amdgpu10.30-amd-amdpal < %s | FileCheck %s --check-prefixes=CHECK,PAL
+
+; gfx1030 allocates LDS in 1024-byte blocks but encodes it in 512-byte units.
+; 8252 bytes occupies 9216 bytes of LDS for occupancy, while its encoded size is
+; 17 units (8704 bytes). Using the allocation granule for either encoding step
+; would produce an incorrect block count or PAL metadata size.
+
+ at lds = addrspace(3) global [2063 x i32] poison, align 4
+
+; CHECK-LABEL: lds_granularity:
+; MESA: granulated_lds_size = 17
+; CHECK: ; Occupancy: 7{{$}}
+; PAL: .hardware_stages:
+; PAL: .cs:
+; PAL: .lds_size: 0x2200
+define amdgpu_kernel void @lds_granularity(i32 %index, i32 %value) #0 {
+ %ptr = getelementptr [2063 x i32], ptr addrspace(3) @lds, i32 0, i32 %index
+ store volatile i32 %value, ptr addrspace(3) %ptr
+ ret void
+}
+
+attributes #0 = { "amdgpu-flat-work-group-size"="1,64" }
+
+!amdgpu.pal.metadata.msgpack = !{!0}
+!0 = !{!"\81\AEamdpal.version\92\03\00"}
diff --git a/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll b/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll
new file mode 100644
index 0000000000000..c02e454ee10b3
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/lds-size-gfx9-4-generic.ll
@@ -0,0 +1,34 @@
+; RUN: llc -mtriple=amdgpu9.42-mesa-mesa3d < %s | FileCheck %s --check-prefix=GFX942
+; RUN: llc -mtriple=amdgpu9.50-mesa-mesa3d < %s | FileCheck %s --check-prefix=GFX950
+; RUN: llc -mtriple=amdgpu9.4-mesa-mesa3d < %s | FileCheck %s --check-prefix=GENERIC
+
+; gfx9-4-generic uses gfx950's 1280-byte allocation and encoding granules
+; independently of its 64 KiB addressable LDS capacity. gfx942 uses 512 bytes
+; for both granularities.
+
+ at lds1280 = addrspace(3) global [320 x i32] poison, align 4
+ at lds1284 = addrspace(3) global [321 x i32] poison, align 4
+
+; GFX942-LABEL: one_granule:
+; GFX942: granulated_lds_size = 3
+; GFX950-LABEL: one_granule:
+; GFX950: granulated_lds_size = 1
+; GENERIC-LABEL: one_granule:
+; GENERIC: granulated_lds_size = 1
+define amdgpu_kernel void @one_granule(i32 %index, i32 %value) {
+ %ptr = getelementptr [320 x i32], ptr addrspace(3) @lds1280, i32 0, i32 %index
+ store volatile i32 %value, ptr addrspace(3) %ptr
+ ret void
+}
+
+; GFX942-LABEL: two_granules:
+; GFX942: granulated_lds_size = 3
+; GFX950-LABEL: two_granules:
+; GFX950: granulated_lds_size = 2
+; GENERIC-LABEL: two_granules:
+; GENERIC: granulated_lds_size = 2
+define amdgpu_kernel void @two_granules(i32 %index, i32 %value) {
+ %ptr = getelementptr [321 x i32], ptr addrspace(3) @lds1284, i32 0, i32 %index
+ store volatile i32 %value, ptr addrspace(3) %ptr
+ ret void
+}
diff --git a/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td b/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
index 13136a7d00b59..b5e4337081a7d 100644
--- a/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
+++ b/llvm/test/TableGen/AMDGPUTargetDefLDSAllocGranularity.td
@@ -3,6 +3,8 @@
// RUN: | FileCheck %t/valid.td
// RUN: not llvm-tblgen -gen-amdgpu-target-def -I %t -I %p/../../include -I %p/../../lib/Target/AMDGPU %t/missing.td 2>&1 \
// RUN: | FileCheck %t/missing.td -DFILE=%t/missing.td --implicit-check-not="error:"
+// RUN: not llvm-tblgen -gen-amdgpu-target-def -I %t -I %p/../../include -I %p/../../lib/Target/AMDGPU %t/missing-encoding.td 2>&1 \
+// RUN: | FileCheck %t/missing-encoding.td -DFILE=%t/missing-encoding.td --implicit-check-not="error:"
//--- common.td
include "llvm/Target/Target.td"
@@ -12,25 +14,42 @@ def MyTarget : Target;
def AMDGPUFrontendVisibleFeatures {
list<SubtargetFeature> Features = [FeatureLDSAllocGranularity512,
FeatureLDSAllocGranularity1024,
- FeatureLDSAllocGranularity1280];
+ FeatureLDSAllocGranularity1280,
+ FeatureLDSEncodingGranularity512,
+ FeatureLDSEncodingGranularity1024,
+ FeatureLDSEncodingGranularity1280];
}
def GFX942 : AMDGPUProcessorModel<"gfx942", NoSchedModel,
- [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512],
+ [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity512,
+ FeatureLDSEncodingGranularity512],
[9, 4, 2]>;
def GFX950 : AMDGPUProcessorModel<"gfx950", NoSchedModel,
- [FeatureAddressableLocalMemorySize163840, FeatureLDSAllocGranularity1280],
+ [FeatureAddressableLocalMemorySize163840, FeatureLDSAllocGranularity1280,
+ FeatureLDSEncodingGranularity1280],
[9, 5, 0]>;
//--- valid.td
include "common.td"
// Capacity and granularity can each come from a different covered GPU.
def : AMDGPUProcessorModel<"gfx9-4-generic", NoSchedModel,
- [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1280],
+ [FeatureAddressableLocalMemorySize65536, FeatureLDSAllocGranularity1280,
+ FeatureLDSEncodingGranularity1280],
[9, 4, 0]> {
let CoveredGPUs = [GFX942, GFX950];
}
-// CHECK: Triple::AMDGPUSubArch9_4, AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1280}), {9, 4, 0}, [[#]], 10, 65536, 32, 0},
+// CHECK: Triple::AMDGPUSubArch9_4, AMDGPUFeatureBitset({FEAT_LDS_ALLOC_GRANULARITY_1280, FEAT_LDS_ENCODING_GRANULARITY_1280}), {9, 4, 0}, [[#]], 10, 65536, 32, 0},
+
+// ...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/224852
More information about the llvm-branch-commits
mailing list