[llvm] bde886d - AMDGPU: Start using subarch in attributor instead of subtarget (#212207)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 10 03:28:25 PDT 2026
Author: Matt Arsenault
Date: 2026-09-10T12:28:21+02:00
New Revision: bde886d4e2a244045390da84a8cb9238ad1d3ab7
URL: https://github.com/llvm/llvm-project/commit/bde886d4e2a244045390da84a8cb9238ad1d3ab7
DIFF: https://github.com/llvm/llvm-project/commit/bde886d4e2a244045390da84a8cb9238ad1d3ab7.diff
LOG: AMDGPU: Start using subarch in attributor instead of subtarget (#212207)
Avoid querying the subtarget for functions when the relevant
properties are known from the triple. The various subtarget
group size functions should also be decoupled from the subtarget,
but those are trickier to untangle.
Co-authored-by: Claude (Opus 4.8) <noreply at anthropic.com>
Added:
Modified:
llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
Removed:
################################################################################
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
index 875657a556329..d3c6505cc23d2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAttributor.cpp
@@ -153,6 +153,8 @@ class AMDGPUInformationCache : public InformationCache {
BumpPtrAllocator &Allocator,
SetVector<Function *> *CGSCC, TargetMachine &TM)
: InformationCache(M, AG, Allocator, CGSCC), TM(TM),
+ Features(AMDGPU::getFeatureBitset(
+ AMDGPU::getGPUKindFromSubArch(M.getTargetTriple().getSubArch()))),
CodeObjectVersion(AMDGPU::getAMDHSACodeObjectVersion(M)) {}
TargetMachine &TM;
@@ -167,18 +169,6 @@ class AMDGPUInformationCache : public InformationCache {
CS_WORST = DS_GLOBAL | ADDR_SPACE_CAST_BOTH_TO_FLAT,
};
- /// Check if the subtarget has aperture regs.
- bool hasApertureRegs(Function &F) {
- const GCNSubtarget &ST = TM.getSubtarget<GCNSubtarget>(F);
- return ST.hasApertureRegs();
- }
-
- /// Check if the subtarget supports GetDoorbellID.
- bool supportsGetDoorbellID(Function &F) {
- const GCNSubtarget &ST = TM.getSubtarget<GCNSubtarget>(F);
- return ST.supportsGetDoorbellID();
- }
-
std::optional<std::pair<unsigned, unsigned>>
getFlatWorkGroupSizeAttr(const Function &F) const {
auto R = AMDGPU::getIntegerPairAttribute(F, "amdgpu-flat-work-group-size");
@@ -202,6 +192,9 @@ class AMDGPUInformationCache : public InformationCache {
/// Get code object version.
unsigned getCodeObjectVersion() const { return CodeObjectVersion; }
+ /// Get the features of the module target.
+ const AMDGPU::AMDGPUFeatureBitset &getFeatures() const { return Features; }
+
std::optional<std::pair<unsigned, unsigned>>
getWavesPerEUAttr(const Function &F) {
auto Val = AMDGPU::getIntegerPairAttribute(F, "amdgpu-waves-per-eu",
@@ -288,7 +281,7 @@ class AMDGPUInformationCache : public InformationCache {
/// Returns true if \p Fn needs the queue pointer because of \p C.
bool needsQueuePtr(const Constant *C, Function &Fn) {
bool IsNonEntryFunc = !AMDGPU::isEntryFunctionCC(Fn.getCallingConv());
- bool HasAperture = hasApertureRegs(Fn);
+ bool HasAperture = Features.test(AMDGPU::FEAT_APERTURE_REGS);
// No need to explore the constants.
if (!IsNonEntryFunc && HasAperture)
@@ -311,6 +304,7 @@ class AMDGPUInformationCache : public InformationCache {
private:
/// Used to determine if the Constant needs the queue pointer.
DenseMap<const Constant *, std::optional<uint8_t>> ConstantStatus;
+ const AMDGPU::AMDGPUFeatureBitset Features;
const unsigned CodeObjectVersion;
};
@@ -503,8 +497,9 @@ struct AAAMDAttributesFunction : public AAAMDAttributes {
bool NeedsImplicit = false;
auto &InfoCache = static_cast<AMDGPUInformationCache &>(A.getInfoCache());
- bool HasApertureRegs = InfoCache.hasApertureRegs(*F);
- bool SupportsGetDoorbellID = InfoCache.supportsGetDoorbellID(*F);
+ const AMDGPU::AMDGPUFeatureBitset &Features = InfoCache.getFeatures();
+ bool HasApertureRegs = Features.test(AMDGPU::FEAT_APERTURE_REGS);
+ bool SupportsGetDoorbellID = Features.test(AMDGPU::FEAT_GET_DOORBELL_ID);
unsigned COV = InfoCache.getCodeObjectVersion();
for (Function *Callee : AAEdges->getOptimisticEdges()) {
@@ -639,7 +634,8 @@ struct AAAMDAttributesFunction : public AAAMDAttributes {
return true;
};
- bool HasApertureRegs = InfoCache.hasApertureRegs(*F);
+ bool HasApertureRegs =
+ InfoCache.getFeatures().test(AMDGPU::FEAT_APERTURE_REGS);
// `checkForAllInstructions` is much more cheaper than going through all
// instructions, try it first.
@@ -1613,11 +1609,11 @@ static bool runImpl(SetVector<Function *> &Functions, bool IsModulePass,
A.getOrCreateAAFor<AAAMDWavesPerEU>(IRPosition::function(*F));
}
- const GCNSubtarget &ST = TM.getSubtarget<GCNSubtarget>(*F);
- if (!F->isDeclaration() && ST.hasClusters())
+ const AMDGPU::AMDGPUFeatureBitset &Features = InfoCache.getFeatures();
+ if (!F->isDeclaration() && Features.test(AMDGPU::FEAT_CLUSTERS))
A.getOrCreateAAFor<AAAMDGPUClusterDims>(IRPosition::function(*F));
- if (ST.hasGFX90AInsts())
+ if (Features.test(AMDGPU::FEAT_AGPR_ALLOC))
A.getOrCreateAAFor<AAAMDGPUMinAGPRAlloc>(IRPosition::function(*F));
for (auto &I : instructions(F)) {
More information about the llvm-commits
mailing list