[llvm] AMDGPU: Query waves-per-EU and flat-wg range from subarch in attributor (PR #222596)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 10 03:40:37 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Matt Arsenault (arsenm)
<details>
<summary>Changes</summary>
The maximum flat work group range is a pair of constants and the maximum
waves per execution unit is fully known from the subarch. Use TargetParser
information and continue working to remove the dependence on codegen.
Co-authored-by: Claude (Opus 4.8) <noreply@<!-- -->anthropic.com>
---
Full diff: https://github.com/llvm/llvm-project/pull/222596.diff
1 Files Affected:
- (modified) llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp (+15-19)
``````````diff
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
index d3c6505cc23d2..2ed1be7c2ade0 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
@@ -153,8 +153,9 @@ class AMDGPUInformationCache : public InformationCache {
BumpPtrAllocator &Allocator,
SetVector<Function *> *CGSCC, TargetMachine &TM)
: InformationCache(M, AG, Allocator, CGSCC), TM(TM),
- Features(AMDGPU::getFeatureBitset(
- AMDGPU::getGPUKindFromSubArch(M.getTargetTriple().getSubArch()))),
+ SubArch(M.getTargetTriple().getSubArch()),
+ Features(
+ AMDGPU::getFeatureBitset(AMDGPU::getGPUKindFromSubArch(SubArch))),
CodeObjectVersion(AMDGPU::getAMDHSACodeObjectVersion(M)) {}
TargetMachine &TM;
@@ -183,10 +184,9 @@ class AMDGPUInformationCache : public InformationCache {
return ST.getDefaultFlatWorkGroupSize(F.getCallingConv());
}
- std::pair<unsigned, unsigned>
- getMaximumFlatWorkGroupRange(const Function &F) {
- const GCNSubtarget &ST = TM.getSubtarget<GCNSubtarget>(F);
- return {ST.getMinFlatWorkGroupSize(), ST.getMaxFlatWorkGroupSize()};
+ std::pair<unsigned, unsigned> getMaximumFlatWorkGroupRange() const {
+ return {AMDGPU::getMinFlatWorkGroupSize(),
+ AMDGPU::getMaxFlatWorkGroupSize()};
}
/// Get code object version.
@@ -201,16 +201,13 @@ class AMDGPUInformationCache : public InformationCache {
/*OnlyFirstRequired=*/true);
if (!Val)
return std::nullopt;
- if (!Val->second) {
- const GCNSubtarget &ST = TM.getSubtarget<GCNSubtarget>(F);
- Val->second = ST.getMaxWavesPerEU();
- }
+ if (!Val->second)
+ Val->second = AMDGPU::getMaxWavesPerEU(SubArch);
return std::make_pair(Val->first, *(Val->second));
}
- unsigned getMaxWavesPerEU(const Function &F) {
- const GCNSubtarget &ST = TM.getSubtarget<GCNSubtarget>(F);
- return ST.getMaxWavesPerEU();
+ unsigned getMaxWavesPerEU() const {
+ return AMDGPU::getMaxWavesPerEU(SubArch);
}
unsigned getMaxAddrSpace() const override {
@@ -304,6 +301,7 @@ class AMDGPUInformationCache : public InformationCache {
private:
/// Used to determine if the Constant needs the queue pointer.
DenseMap<const Constant *, std::optional<uint8_t>> ConstantStatus;
+ const Triple::SubArchType SubArch;
const AMDGPU::AMDGPUFeatureBitset Features;
const unsigned CodeObjectVersion;
};
@@ -879,7 +877,7 @@ struct AAAMDFlatWorkGroupSize : public AAAMDSizeRangeAttribute {
bool HasAttr = false;
auto Range = InfoCache.getDefaultFlatWorkGroupSize(*F);
- auto MaxRange = InfoCache.getMaximumFlatWorkGroupRange(*F);
+ auto MaxRange = InfoCache.getMaximumFlatWorkGroupRange();
if (auto Attr = InfoCache.getFlatWorkGroupSizeAttr(*F)) {
// We only consider an attribute that is not max range because the front
@@ -914,10 +912,9 @@ struct AAAMDFlatWorkGroupSize : public AAAMDSizeRangeAttribute {
Attributor &A);
ChangeStatus manifest(Attributor &A) override {
- Function *F = getAssociatedFunction();
auto &InfoCache = static_cast<AMDGPUInformationCache &>(A.getInfoCache());
return emitAttributeIfNotDefaultAfterClamp(
- A, InfoCache.getMaximumFlatWorkGroupRange(*F));
+ A, InfoCache.getMaximumFlatWorkGroupRange());
}
/// See AbstractAttribute::getName()
@@ -1097,7 +1094,7 @@ struct AAAMDWavesPerEU : public AAAMDSizeRangeAttribute {
// If the attribute exists, we will honor it if it is not the default.
if (auto Attr = InfoCache.getWavesPerEUAttr(*F)) {
std::pair<unsigned, unsigned> MaxWavesPerEURange{
- 1U, InfoCache.getMaxWavesPerEU(*F)};
+ 1U, InfoCache.getMaxWavesPerEU()};
if (*Attr != MaxWavesPerEURange) {
auto [Min, Max] = *Attr;
ConstantRange Range(APInt(32, Min), APInt(32, Max + 1));
@@ -1153,10 +1150,9 @@ struct AAAMDWavesPerEU : public AAAMDSizeRangeAttribute {
Attributor &A);
ChangeStatus manifest(Attributor &A) override {
- Function *F = getAssociatedFunction();
auto &InfoCache = static_cast<AMDGPUInformationCache &>(A.getInfoCache());
return emitAttributeIfNotDefaultAfterClamp(
- A, {1U, InfoCache.getMaxWavesPerEU(*F)});
+ A, {1U, InfoCache.getMaxWavesPerEU()});
}
/// See AbstractAttribute::getName()
``````````
</details>
https://github.com/llvm/llvm-project/pull/222596
More information about the llvm-commits
mailing list