[llvm] Codegen - AMDGPU Subtarget FIXMES (PR #205499)
Nikhil Kotikalapudi via llvm-commits
llvm-commits at lists.llvm.org
Wed Jun 24 01:11:09 PDT 2026
https://github.com/23silicon updated https://github.com/llvm/llvm-project/pull/205499
>From cb79ba408f394d26701a14e9d01f97fd3b69af93 Mon Sep 17 00:00:00 2001
From: Nikhil Kotikalapudi <Nikhil.Kotikalapudi at amd.com>
Date: Wed, 24 Jun 2026 01:41:25 -0500
Subject: [PATCH 1/2] relocated getMaxNumWorkGroups function and d refactored
---
llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp | 3 +--
llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp | 7 -------
llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h | 3 ---
llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp | 4 ++--
llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp | 2 +-
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp | 6 ++++++
llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h | 3 +++
7 files changed, 13 insertions(+), 15 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
index 0ddbb92783c39..7ebcf2ca47cdc 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
@@ -200,8 +200,7 @@ class AMDGPUInformationCache : public InformationCache {
}
SmallVector<unsigned> getMaxNumWorkGroups(const Function &F) {
- const GCNSubtarget &ST = TM.getSubtarget<GCNSubtarget>(F);
- return ST.getMaxNumWorkGroups(F);
+ return AMDGPU::getMaxNumWorkGroups(F);
}
/// Get code object version.
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
index fb4728609c877..3122eb60e160c 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
@@ -430,10 +430,3 @@ const AMDGPUSubtarget &AMDGPUSubtarget::get(const TargetMachine &TM, const Funct
return static_cast<const AMDGPUSubtarget &>(
TM.getSubtarget<R600Subtarget>(F));
}
-
-// FIXME: This has no reason to be in subtarget
-SmallVector<unsigned>
-AMDGPUSubtarget::getMaxNumWorkGroups(const Function &F) const {
- return AMDGPU::getIntegerVecAttribute(F, "amdgpu-max-num-workgroups", 3,
- std::numeric_limits<uint32_t>::max());
-}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
index 07746c087904d..f1ff6d80a55b3 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
@@ -294,9 +294,6 @@ class AMDGPUSubtarget {
/// 2) dimension.
unsigned getMaxWorkitemID(const Function &Kernel, unsigned Dimension) const;
- /// Return the number of work groups for the function.
- SmallVector<unsigned> getMaxNumWorkGroups(const Function &F) const;
-
/// Return true if only a single workitem can be active in a wave.
bool isSingleLaneExecution(const Function &Kernel) const;
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
index fe66a1a5d7242..15f2598ae9e92 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp
@@ -1781,9 +1781,9 @@ bool GCNTTIImpl::shouldPrefetchAddressSpace(unsigned AS) const {
void GCNTTIImpl::collectKernelLaunchBounds(
const Function &F,
SmallVectorImpl<std::pair<StringRef, int64_t>> &LB) const {
- SmallVector<unsigned> MaxNumWorkgroups = ST->getMaxNumWorkGroups(F);
+ SmallVector<unsigned> MaxNumWorkgroups = AMDGPU::getMaxNumWorkGroups(F);
LB.push_back({"amdgpu-max-num-workgroups[0]", MaxNumWorkgroups[0]});
- LB.push_back({"amdgpu-max-num-workgroups[1]", MaxNumWorkgroups[1]});
+ LB.push_back({"amdgpu-max-num-workgroups[1]", MaxNumWorkgroups[2]});
LB.push_back({"amdgpu-max-num-workgroups[2]", MaxNumWorkgroups[2]});
std::pair<unsigned, unsigned> FlatWorkGroupSize =
ST->getFlatWorkGroupSizes(F);
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
index 4be4ce28e6de5..59971923e4a5d 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.cpp
@@ -60,7 +60,7 @@ SIMachineFunctionInfo::SIMachineFunctionInfo(const Function &F,
const GCNSubtarget &ST = *STI;
FlatWorkGroupSizes = ST.getFlatWorkGroupSizes(F);
WavesPerEU = ST.getWavesPerEU(F);
- MaxNumWorkGroups = ST.getMaxNumWorkGroups(F);
+ MaxNumWorkGroups = AMDGPU::getMaxNumWorkGroups(F);
assert(MaxNumWorkGroups.size() == 3);
// Temporarily check both the attribute and the subtarget feature, until the
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
index 96571dd028b14..f74133fcfc0c2 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.cpp
@@ -1756,6 +1756,12 @@ getIntegerVecAttribute(const Function &F, StringRef Name, unsigned Size) {
return Vals;
}
+SmallVector<unsigned>
+getMaxNumWorkGroups(const Function &F) {
+ return getIntegerVecAttribute(F, "amdgpu-max-num-workgroups", 3,
+ std::numeric_limits<uint32_t>::max());
+}
+
bool hasValueInRangeLikeMetadata(const MDNode &MD, int64_t Val) {
assert((MD.getNumOperands() % 2 == 0) && "invalid number of operands!");
for (unsigned I = 0, E = MD.getNumOperands() / 2; I != E; ++I) {
diff --git a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
index 1623dc72d2810..3c42fb1a31005 100644
--- a/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
+++ b/llvm/lib/Target/AMDGPU/Utils/AMDGPUBaseInfo.h
@@ -1029,6 +1029,9 @@ SmallVector<unsigned> getIntegerVecAttribute(const Function &F, StringRef Name,
std::optional<SmallVector<unsigned>>
getIntegerVecAttribute(const Function &F, StringRef Name, unsigned Size);
+/// \returns The maximum number of workgroups for the function.
+SmallVector<unsigned> getMaxNumWorkGroups(const Function &F);
+
/// Checks if \p Val is inside \p MD, a !range-like metadata.
bool hasValueInRangeLikeMetadata(const MDNode &MD, int64_t Val);
>From e004d9ba6d15ad78385ff5b52913b4e1f72d2213 Mon Sep 17 00:00:00 2001
From: Nikhil Kotikalapudi <Nikhil.Kotikalapudi at amd.com>
Date: Wed, 24 Jun 2026 02:46:54 -0500
Subject: [PATCH 2/2] Align LDS request to granularity for MaxWGsLDS
calculation Test, GFX1250 expected occupancy hardcode bugfix
---
llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp | 5 ++++-
llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h | 1 +
llvm/lib/Target/AMDGPU/GCNSubtarget.cpp | 3 +++
llvm/test/CodeGen/AMDGPU/occupancy-levels.ll | 2 +-
4 files changed, 9 insertions(+), 2 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
index 3122eb60e160c..93bf957880074 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.cpp
@@ -52,7 +52,10 @@ AMDGPUSubtarget::getMaxLocalMemSizeWithWaveCount(unsigned NWaves,
std::pair<unsigned, unsigned> AMDGPUSubtarget::getOccupancyWithWorkGroupSizes(
uint32_t LDSBytes, std::pair<unsigned, unsigned> FlatWorkGroupSizes) const {
- // FIXME: We should take into account the LDS allocation granularity.
+ // LDS granularity accounted for by aligning the queried LDS size to the
+ // allocation block size.
+ const unsigned Granularity = std::max(LDSAllocationGranularity, 1u);
+ LDSBytes = alignTo(LDSBytes, Granularity);
const unsigned MaxWGsLDS = getLocalMemorySize() / std::max(LDSBytes, 1u);
// Queried LDS size may be larger than available on a CU, in which case we
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
index f1ff6d80a55b3..af3facc0135f0 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUSubtarget.h
@@ -58,6 +58,7 @@ class AMDGPUSubtarget {
unsigned MaxWavesPerEU = 10;
unsigned LocalMemorySize = 0;
unsigned AddressableLocalMemorySize = 0;
+ unsigned LDSAllocationGranularity = 0;
char WavefrontSizeLog2 = 0;
unsigned FlatOffsetBitWidth = 0;
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
index 37efb3a51cb9d..23c92b9095e36 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
@@ -144,6 +144,9 @@ GCNSubtarget &GCNSubtarget::initializeSubtargetDependencies(const Triple &TT,
FlatOffsetBitWidth = 13;
LocalMemorySize = AMDGPU::IsaInfo::getLocalMemorySize(*this);
+ // LDS Allocation Granularity calculated in bytes from dwords
+ LDSAllocationGranularity =
+ AMDGPU::getLdsDwGranularity(*this) * sizeof(uint32_t);
HasFminFmaxLegacy = getGeneration() < AMDGPUSubtarget::VOLCANIC_ISLANDS;
HasSMulHi = getGeneration() >= AMDGPUSubtarget::GFX9;
diff --git a/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll b/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll
index 2ede4248508ed..d3251d3c8bc4a 100644
--- a/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll
+++ b/llvm/test/CodeGen/AMDGPU/occupancy-levels.ll
@@ -460,7 +460,7 @@ define amdgpu_kernel void @used_lds_13112() {
; GFX10W32: ; Occupancy: 8{{$}}
; GFX1100W64: ; Occupancy: 4{{$}}
; GFX1100W32: ; Occupancy: 8{{$}}
-; GFX1250: ; Occupancy: 10{{$}}
+; GFX1250: ; Occupancy: 8{{$}}
@lds8252 = internal addrspace(3) global [8252 x i8] poison, align 4
define amdgpu_kernel void @used_lds_8252_max_group_size_64() #3 {
store volatile i8 1, ptr addrspace(3) @lds8252
More information about the llvm-commits
mailing list