[llvm] [AMDGPU] Preserve resource requirements through call-graph cycles (PR #221625)
via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 13 21:58:47 PDT 2026
https://github.com/mmiftahx updated https://github.com/llvm/llvm-project/pull/221625
>From e045f0f45a305e62fc20362dcd4bf811de9b1e0e Mon Sep 17 00:00:00 2001
From: mmiftahx <mmiftah.duna at gmail.com>
Date: Sun, 6 Sep 2026 22:14:43 -0500
Subject: [PATCH 1/2] [AMDGPU] Preserve resource requirements through
call-graph cycles
Retain unresolved outgoing callees when normalizing cyclic max/OR resource
expressions. Preserve register and boolean requirements through the cycle,
including for dynamically acyclic paths annotated norecurse.
Fold constants and duplicate contributions across the complete assignment.
Evaluate shared max/OR/addition expressions once per query, preserving symbol
resolution and first-invalid-operand behavior.
Propagate a recursion resource flag when the private-segment expression drops
a cyclic edge. Emit the dynamic-stack descriptor bit only for code object V5
and later; this does not establish a finite bound for older code objects.
Add direct-call, expression, and V4/V5 object-emission regressions. Cover both
instruction selectors, all six graph emission orders, deduplicated shared
outgoing edges, and a shared 64-function private-stack DAG with an exact 1024-byte bound.
---
llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp | 8 +-
.../Target/AMDGPU/AMDGPUMCResourceInfo.cpp | 193 +++++++++---------
llvm/lib/Target/AMDGPU/AMDGPUMCResourceInfo.h | 12 +-
.../AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 190 ++++++++++++++---
.../CodeGen/AMDGPU/function-resource-usage.ll | 14 +-
.../AMDGPU/recursive-resource-usage-mcexpr.ll | 20 +-
.../resource-usage-cycle-shared-outgoing.mir | 52 +++++
.../resource-usage-direct-call-cycle.ll | 76 +++++++
.../resource-usage-direct-call-cycle.mir | 110 ++++++++++
.../AMDGPU/resource-usage-norecurse-cycle.ll | 112 ++++++++++
.../resource-usage-private-stack-dag.test | 61 ++++++
llvm/test/MC/AMDGPU/idempotent-expr-errors.s | 101 +++++++++
.../MC/AMDGPU/idempotent-expr-evaluation.s | 163 +++++++++++++++
13 files changed, 966 insertions(+), 146 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/resource-usage-cycle-shared-outgoing.mir
create mode 100644 llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.ll
create mode 100644 llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.mir
create mode 100644 llvm/test/CodeGen/AMDGPU/resource-usage-norecurse-cycle.ll
create mode 100644 llvm/test/CodeGen/AMDGPU/resource-usage-private-stack-dag.test
create mode 100644 llvm/test/MC/AMDGPU/idempotent-expr-errors.s
create mode 100644 llvm/test/MC/AMDGPU/idempotent-expr-evaluation.s
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
index f754fd8bd24f3..c8f9bc4c747bd 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUAsmPrinter.cpp
@@ -837,12 +837,16 @@ const MCExpr *AMDGPUAsmPrinter::getAmdhsaKernelCodeProperties(
amdhsa::KERNEL_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32;
}
+ const MCExpr *KernelCodePropExpr =
+ MCConstantExpr::create(KernelCodeProperties, Ctx);
+ // The dynamic-stack bit is reserved in code objects before V5.
+ if (CodeObjectVersion < AMDGPU::AMDHSA_COV5)
+ return KernelCodePropExpr;
+
// CurrentProgramInfo.DynamicCallStack is a MCExpr and could be
// un-evaluatable at this point so it cannot be conditionally checked here.
// Instead, we'll directly shift the possibly unknown MCExpr into its place
// and bitwise-or it into KernelCodeProperties.
- const MCExpr *KernelCodePropExpr =
- MCConstantExpr::create(KernelCodeProperties, Ctx);
const MCExpr *OrValue = MCConstantExpr::create(
amdhsa::KERNEL_CODE_PROPERTY_USES_DYNAMIC_STACK_SHIFT, Ctx);
OrValue = MCBinaryExpr::createShl(CurrentProgramInfo.DynamicCallStack,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCResourceInfo.cpp b/llvm/lib/Target/AMDGPU/AMDGPUMCResourceInfo.cpp
index 1731b0463a099..9dfecb95acb81 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCResourceInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCResourceInfo.cpp
@@ -15,6 +15,7 @@
#include "AMDGPUMCResourceInfo.h"
#include "SIMachineFunctionInfo.h"
#include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/StringRef.h"
#include "llvm/MC/MCAsmInfo.h"
#include "llvm/MC/MCContext.h"
@@ -105,84 +106,86 @@ MCSymbol *MCResourceInfo::getMaxNamedBarrierSymbol(MCContext &OutContext) {
return OutContext.getOrCreateSymbol("amdgpu.max_num_named_barrier");
}
-// Tries to flatten recursive call register resource gathering. Simple cycle
-// avoiding dfs to find the constants in the propagated symbols.
-// Assumes:
-// - RecSym has been confirmed to recurse (this means the callee symbols should
-// all be populated, started at RecSym).
-// - Shape of the resource symbol's MCExpr (`max` args are order agnostic):
-// RecSym.MCExpr := max(<constant>+, <callee_symbol>*)
-const MCExpr *MCResourceInfo::flattenedCycleMax(MCSymbol *RecSym,
- ResourceInfoKind RIK,
- MCContext &OutContext) {
+// Max and OR are associative and idempotent. Normalize the whole cyclic
+// assignment, folding its constants and retaining each unresolved outgoing
+// symbol once, using one visited set across all callees.
+const MCExpr *MCResourceInfo::flattenedCycleExpr(
+ MCSymbol *Sym, const MCExpr *Expr, int64_t LocalValue, ResourceInfoKind RIK,
+ AMDGPUMCExpr::VariantKind Kind, MCContext &OutContext) {
+ assert((Kind == AMDGPUMCExpr::AGVK_Max || Kind == AMDGPUMCExpr::AGVK_Or) &&
+ "expected an idempotent resource expression");
SmallPtrSet<const MCExpr *, 8> Seen;
- SmallVector<const MCExpr *, 8> WorkList;
- int64_t Maximum = 0;
-
- const MCExpr *RecExpr = RecSym->getVariableValue();
- WorkList.push_back(RecExpr);
+ SmallPtrSet<const MCSymbol *, 8> UnresolvedSymbols;
+ SmallVector<const MCExpr *, 8> WorkList{Expr};
+ SmallVector<const MCExpr *, 8> Args{nullptr};
+ int64_t ConstantValue = 0;
while (!WorkList.empty()) {
const MCExpr *CurExpr = WorkList.pop_back_val();
- switch (CurExpr->getKind()) {
- default: {
- // Assuming the recursion is of shape `max(<constant>, <callee_symbol>)`
- // where <callee_symbol> will eventually recurse. If this condition holds,
- // the recursion occurs within some other (possibly unresolvable) MCExpr,
- // thus using the worst case value then.
- if (!AMDGPUMCExpr::isSymbolUsedInExpression(RecSym, CurExpr)) {
- LLVM_DEBUG(dbgs() << "MCResUse: " << RecSym->getName()
- << ": Recursion in unexpected sub-expression, using "
- "module maximum\n");
- switch (RIK) {
- default:
- break;
- case RIK_NumVGPR:
- return MCSymbolRefExpr::create(getMaxVGPRSymbol(OutContext),
- OutContext);
- break;
- case RIK_NumSGPR:
- return MCSymbolRefExpr::create(getMaxSGPRSymbol(OutContext),
- OutContext);
- break;
- case RIK_NumAGPR:
- return MCSymbolRefExpr::create(getMaxAGPRSymbol(OutContext),
- OutContext);
- break;
- }
- }
- break;
+ if (!Seen.insert(CurExpr).second)
+ continue;
+
+ if (const auto *Constant = dyn_cast<MCConstantExpr>(CurExpr)) {
+ ConstantValue = Kind == AMDGPUMCExpr::AGVK_Max
+ ? std::max(ConstantValue, Constant->getValue())
+ : ConstantValue | Constant->getValue();
+ continue;
}
- case MCExpr::ExprKind::Constant: {
- int64_t Val = cast<MCConstantExpr>(CurExpr)->getValue();
- Maximum = std::max(Maximum, Val);
- break;
+
+ if (const auto *SymExpr = dyn_cast<MCSymbolRefExpr>(CurExpr)) {
+ const MCSymbol &Ref = SymExpr->getSymbol();
+ // The current function's local value is already part of the assignment.
+ // A reference to it through a call cycle contributes nothing further.
+ if (&Ref == Sym)
+ continue;
+ if (Ref.isVariable())
+ WorkList.push_back(Ref.getVariableValue());
+ else if (UnresolvedSymbols.insert(&Ref).second)
+ Args.push_back(CurExpr);
+ continue;
}
- case MCExpr::ExprKind::SymbolRef: {
- const MCSymbolRefExpr *SymExpr = cast<MCSymbolRefExpr>(CurExpr);
- const MCSymbol &SymRef = SymExpr->getSymbol();
- if (SymRef.isVariable()) {
- const MCExpr *SymVal = SymRef.getVariableValue();
- if (Seen.insert(SymVal).second)
- WorkList.push_back(SymVal);
+
+ if (const auto *TargetExpr = dyn_cast<AMDGPUMCExpr>(CurExpr)) {
+ if (TargetExpr->getKind() == Kind) {
+ // Push in reverse order to preserve the original expression order.
+ append_range(WorkList, reverse(TargetExpr->getArgs()));
+ continue;
}
- break;
}
- case MCExpr::ExprKind::Target: {
- const AMDGPUMCExpr *TargetExpr = cast<AMDGPUMCExpr>(CurExpr);
- if (TargetExpr->getKind() == AMDGPUMCExpr::VariantKind::AGVK_Max) {
- for (auto &Arg : TargetExpr->getArgs())
- WorkList.push_back(Arg);
- }
+
+ // Resource symbols currently consist only of constants, symbol references,
+ // and max/OR expressions. If that changes, an unfamiliar shape within a
+ // cyclic assignment must still produce a conservative value.
+ LLVM_DEBUG(dbgs() << "MCResUse: " << Sym->getName()
+ << ": Unexpected cyclic expression; using fallback\n");
+ MCSymbol *MaxSym;
+ switch (RIK) {
+ case RIK_NumVGPR:
+ MaxSym = getMaxVGPRSymbol(OutContext);
break;
+ case RIK_NumAGPR:
+ MaxSym = getMaxAGPRSymbol(OutContext);
+ break;
+ case RIK_NumSGPR:
+ MaxSym = getMaxSGPRSymbol(OutContext);
+ break;
+ case RIK_NumNamedBarrier:
+ MaxSym = getMaxNamedBarrierSymbol(OutContext);
+ break;
+ default:
+ return MCConstantExpr::create(1, OutContext);
}
- }
+ // Entry functions do not contribute to the module maximum.
+ return AMDGPUMCExpr::createMax(
+ {MCConstantExpr::create(LocalValue, OutContext),
+ MCSymbolRefExpr::create(MaxSym, OutContext)},
+ OutContext);
}
- LLVM_DEBUG(dbgs() << "MCResUse: " << RecSym->getName()
- << ": Using flattened max: << " << Maximum << '\n');
-
- return MCConstantExpr::create(Maximum, OutContext);
+ Args.front() = MCConstantExpr::create(ConstantValue, OutContext);
+ if (Args.size() == 1)
+ return Args.front();
+ return AMDGPUMCExpr::create(Kind, Args, OutContext);
}
void MCResourceInfo::assignResourceInfoExpr(
@@ -209,39 +212,21 @@ void MCResourceInfo::assignResourceInfoExpr(
MCSymbol *CalleeFnSym = TM.getSymbol(&Callee->getFunction());
MCSymbol *CalleeValSym =
getSymbol(CalleeFnSym->getName(), RIK, OutContext);
-
- // Avoid constructing recursive definitions by detecting whether `Sym` is
- // found transitively within any of its `CalleeValSym`.
- if (!CalleeValSym->isVariable() ||
- !AMDGPUMCExpr::isSymbolUsedInExpression(
- Sym, CalleeValSym->getVariableValue())) {
- LLVM_DEBUG(dbgs() << "MCResUse: " << Sym->getName() << ": Adding "
- << CalleeValSym->getName() << " as callee\n");
- ArgExprs.push_back(MCSymbolRefExpr::create(CalleeValSym, OutContext));
- } else {
+ LLVM_DEBUG(dbgs() << "MCResUse: " << Sym->getName() << ": Adding "
+ << CalleeValSym->getName() << " as callee\n");
+ ArgExprs.push_back(MCSymbolRefExpr::create(CalleeValSym, OutContext));
+ }
+ if (ArgExprs.size() > 1) {
+ SymVal = AMDGPUMCExpr::create(Kind, ArgExprs, OutContext);
+ // Detect the cycle once, then normalize the complete assignment. A
+ // single visited set folds shared subexpressions across every callee.
+ if (AMDGPUMCExpr::isSymbolUsedInExpression(Sym, SymVal)) {
LLVM_DEBUG(dbgs() << "MCResUse: " << Sym->getName()
- << ": Recursion found, attempt flattening of cycle "
- "for resource usage\n");
- // In case of recursion for vgpr/sgpr/agpr resource usage: try to
- // flatten and use the max of the call cycle. May still end up emitting
- // module max if not fully resolvable.
- switch (RIK) {
- default:
- break;
- case RIK_NumVGPR:
- case RIK_NumSGPR:
- case RIK_NumAGPR:
- ArgExprs.push_back(flattenedCycleMax(CalleeValSym, RIK, OutContext));
- break;
- case RIK_NumNamedBarrier:
- ArgExprs.push_back(MCSymbolRefExpr::create(
- getMaxNamedBarrierSymbol(OutContext), OutContext));
- break;
- }
+ << ": Normalizing cyclic resource expression\n");
+ SymVal =
+ flattenedCycleExpr(Sym, SymVal, LocalValue, RIK, Kind, OutContext);
}
}
- if (ArgExprs.size() > 1)
- SymVal = AMDGPUMCExpr::create(Kind, ArgExprs, OutContext);
}
Sym->setVariableValue(SymVal);
}
@@ -353,6 +338,7 @@ void MCResourceInfo::gatherResourceInfo(
SetMaxReg(MaxSGPRSym, FRI.NumExplicitSGPR, RIK_NumSGPR);
SetMaxReg(MaxNamedBarrierSym, FRI.NumNamedBarrier, RIK_NumNamedBarrier);
+ bool HasPrivateSegmentCycle = false;
{
// The expression for private segment size should be: FRI.PrivateSegmentSize
// + max(FRI.Callees, FRI.CalleeSegmentSize)
@@ -384,6 +370,15 @@ void MCResourceInfo::gatherResourceInfo(
LLVM_DEBUG(dbgs() << "MCResUse: " << Sym->getName() << ": Adding "
<< CalleeValSym->getName() << " as callee\n");
ArgExprs.push_back(MCSymbolRefExpr::create(CalleeValSym, OutContext));
+ } else {
+ // Dropping this edge avoids a recursive MC expression, but it also
+ // means the finite private-segment expression is no longer a proven
+ // call-graph closure. Request dynamic-stack provisioning on versions
+ // that support it. This does not prove a bound for older versions.
+ HasPrivateSegmentCycle = true;
+ LLVM_DEBUG(dbgs() << "MCResUse: " << Sym->getName()
+ << ": Private-segment call cycle found; using "
+ "dynamic-stack fallback\n");
}
}
}
@@ -410,7 +405,8 @@ void MCResourceInfo::gatherResourceInfo(
assignResourceInfoExpr(FRI.HasDynamicallySizedStack,
ResourceInfoKind::RIK_HasDynSizedStack,
AMDGPUMCExpr::AGVK_Or, MF, FRI.Callees, OutContext);
- assignResourceInfoExpr(FRI.HasRecursion, ResourceInfoKind::RIK_HasRecursion,
+ assignResourceInfoExpr(FRI.HasRecursion || HasPrivateSegmentCycle,
+ ResourceInfoKind::RIK_HasRecursion,
AMDGPUMCExpr::AGVK_Or, MF, FRI.Callees, OutContext);
assignResourceInfoExpr(FRI.HasIndirectCall,
ResourceInfoKind::RIK_HasIndirectCall,
@@ -420,7 +416,8 @@ void MCResourceInfo::gatherResourceInfo(
SetToLocal(FRI.UsesFlatScratch, ResourceInfoKind::RIK_UsesFlatScratch);
SetToLocal(FRI.HasDynamicallySizedStack,
ResourceInfoKind::RIK_HasDynSizedStack);
- SetToLocal(FRI.HasRecursion, ResourceInfoKind::RIK_HasRecursion);
+ SetToLocal(FRI.HasRecursion || HasPrivateSegmentCycle,
+ ResourceInfoKind::RIK_HasRecursion);
SetToLocal(FRI.HasIndirectCall, ResourceInfoKind::RIK_HasIndirectCall);
}
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUMCResourceInfo.h b/llvm/lib/Target/AMDGPU/AMDGPUMCResourceInfo.h
index 3cb063ef8e962..2ac0856cb6fa6 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUMCResourceInfo.h
+++ b/llvm/lib/Target/AMDGPU/AMDGPUMCResourceInfo.h
@@ -60,11 +60,13 @@ class MCResourceInfo {
// Assigns expression for Max S/V/A-GPRs to the referenced symbols.
void assignMaxRegs(MCContext &OutContext);
- // Take flattened max of cyclic function calls' knowns. For example, for
- // a cycle A->B->C->D->A, take max(A, B, C, D) for A and have B, C, D have the
- // propgated value from A.
- const MCExpr *flattenedCycleMax(MCSymbol *RecSym, ResourceInfoKind RIK,
- MCContext &OutContext);
+ // Normalize a whole cyclic max/OR assignment. Fold constants, omit the
+ // symbol currently being assigned, and keep each unresolved outgoing symbol
+ // once. LocalValue is retained even if an unfamiliar shape needs a fallback.
+ const MCExpr *flattenedCycleExpr(MCSymbol *Sym, const MCExpr *Expr,
+ int64_t LocalValue, ResourceInfoKind RIK,
+ AMDGPUMCExpr::VariantKind Kind,
+ MCContext &OutContext);
public:
MCResourceInfo() = default;
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 745647e235f6e..86459a70063e8 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -8,6 +8,11 @@
#include "AMDGPUMCExpr.h"
#include "Utils/AMDGPUBaseInfo.h"
+#include "llvm/ADT/DenseMap.h"
+#include "llvm/ADT/STLExtras.h"
+#include "llvm/ADT/ScopeExit.h"
+#include "llvm/ADT/SmallPtrSet.h"
+#include "llvm/ADT/SmallVector.h"
#include "llvm/MC/MCAsmInfo.h"
#include "llvm/MC/MCAssembler.h"
#include "llvm/MC/MCContext.h"
@@ -247,33 +252,167 @@ bool AMDGPUMCExpr::evaluateInstPrefSize(MCValue &Res,
bool AMDGPUMCExpr::isSymbolUsedInExpression(const MCSymbol *Sym,
const MCExpr *E) {
- switch (E->getKind()) {
- case MCExpr::Constant:
- return false;
- case MCExpr::Unary:
- return isSymbolUsedInExpression(
- Sym, static_cast<const MCUnaryExpr *>(E)->getSubExpr());
- case MCExpr::Binary: {
- const MCBinaryExpr *BE = static_cast<const MCBinaryExpr *>(E);
- return isSymbolUsedInExpression(Sym, BE->getLHS()) ||
- isSymbolUsedInExpression(Sym, BE->getRHS());
- }
- case MCExpr::SymbolRef: {
- const MCSymbol &S = static_cast<const MCSymbolRefExpr *>(E)->getSymbol();
- if (S.isVariable())
- return isSymbolUsedInExpression(Sym, S.getVariableValue());
- return &S == Sym;
- }
- case MCExpr::Specifier:
- case MCExpr::Target: {
- auto *TE = static_cast<const AMDGPUMCExpr *>(E);
- for (const MCExpr *E : TE->getArgs())
- if (isSymbolUsedInExpression(Sym, E))
+ SmallVector<const MCExpr *, 16> WorkList{E};
+ SmallPtrSet<const MCExpr *, 16> Seen;
+ while (!WorkList.empty()) {
+ const MCExpr *Expr = WorkList.pop_back_val();
+ if (!Seen.insert(Expr).second)
+ continue;
+ switch (Expr->getKind()) {
+ case MCExpr::Constant:
+ break;
+ case MCExpr::Unary:
+ WorkList.push_back(cast<MCUnaryExpr>(Expr)->getSubExpr());
+ break;
+ case MCExpr::Binary: {
+ const auto *BE = cast<MCBinaryExpr>(Expr);
+ WorkList.push_back(BE->getRHS());
+ WorkList.push_back(BE->getLHS());
+ break;
+ }
+ case MCExpr::SymbolRef: {
+ const MCSymbol &S = cast<MCSymbolRefExpr>(Expr)->getSymbol();
+ if (&S == Sym)
return true;
- return false;
+ if (S.isVariable())
+ WorkList.push_back(S.getVariableValue());
+ break;
+ }
+ case MCExpr::Specifier:
+ WorkList.push_back(cast<MCSpecifierExpr>(Expr)->getSubExpr());
+ break;
+ case MCExpr::Target:
+ append_range(WorkList, reverse(cast<AMDGPUMCExpr>(Expr)->getArgs()));
+ break;
+ }
}
+ return false;
+}
+
+// Resource expressions form a DAG of maxima, boolean unions, and frame-size
+// additions. Evaluate shared subexpressions once per query, including additions
+// between maxima, instead of recursively expanding every path through the DAG.
+static bool evaluateResourceExpr(const AMDGPUMCExpr *Root, MCValue &Res,
+ const MCAssembler *Asm) {
+ enum class Phase { Visit, Complete, CheckAbsolute };
+ struct WorkItem {
+ const MCExpr *Expr;
+ Phase Step = Phase::Visit;
+ MCSymbol *ResolvingSymbol = nullptr;
+ };
+ SmallVector<WorkItem, 16> WorkList{{Root}};
+ DenseMap<const MCExpr *, MCValue> Values;
+ SmallPtrSet<const MCExpr *, 16> Active;
+ SmallVector<MCSymbol *, 8> ResolvingSymbols;
+ // Match MCExpr's resolution guard, including on failure. Cache values only
+ // for this query: symbol definitions and assembler layout can change later.
+ auto ClearResolving = scope_exit([&] {
+ for (MCSymbol *Sym : ResolvingSymbols)
+ Sym->setIsResolving(false);
+ });
+ auto EvaluateLeaf = [&](const MCExpr *Expr) {
+ MCValue Value;
+ if (!Expr->evaluateAsRelocatable(Value, Asm))
+ return false;
+ Values.try_emplace(Expr, Value);
+ return true;
+ };
+
+ while (!WorkList.empty()) {
+ WorkItem Item = WorkList.pop_back_val();
+ const MCExpr *Expr = Item.Expr;
+ if (Item.Step == Phase::CheckAbsolute) {
+ // Match normal max/OR evaluation: stop before the next operand if this
+ // operand is not absolute.
+ if (!Values.lookup(Expr).isAbsolute())
+ return false;
+ continue;
+ }
+ if (Item.Step == Phase::Complete) {
+ Active.erase(Expr);
+ if (MCSymbol *Sym = Item.ResolvingSymbol) {
+ MCValue Value = Values.lookup(Sym->getVariableValue());
+ Sym->setIsResolving(false);
+ if (Value.isAbsolute())
+ Values.try_emplace(Expr, Value);
+ // The normal resolver preserves the identity of relocatable aliases.
+ // Clear our guard first so it does not diagnose a spurious cycle.
+ else if (!EvaluateLeaf(Expr))
+ return false;
+ continue;
+ }
+ if (const auto *Binary = dyn_cast<MCBinaryExpr>(Expr)) {
+ MCValue LHS = Values.lookup(Binary->getLHS());
+ MCValue RHS = Values.lookup(Binary->getRHS());
+ if (LHS.isAbsolute() && RHS.isAbsolute()) {
+ // Match the generic evaluator's wrapping addition.
+ uint64_t Sum =
+ uint64_t(LHS.getConstant()) + uint64_t(RHS.getConstant());
+ Values.try_emplace(Expr, MCValue::get(Sum));
+ } else if (!EvaluateLeaf(Expr)) {
+ // Let MC handle relocations and layout-dependent symbol arithmetic.
+ return false;
+ }
+ continue;
+ }
+ const auto *Target = cast<AMDGPUMCExpr>(Expr);
+ std::optional<int64_t> Total;
+ for (const MCExpr *Arg : Target->getArgs()) {
+ MCValue Value = Values.lookup(Arg);
+ assert(Value.isAbsolute() && "operand passed CheckAbsolute");
+ Total = Total ? op(Target->getKind(), *Total, Value.getConstant())
+ : Value.getConstant();
+ }
+ Values.try_emplace(Expr, MCValue::get(*Total));
+ continue;
+ }
+ if (Values.contains(Expr))
+ continue;
+
+ if (const auto *Target = dyn_cast<AMDGPUMCExpr>(Expr);
+ Target && (Target->getKind() == AMDGPUMCExpr::AGVK_Max ||
+ Target->getKind() == AMDGPUMCExpr::AGVK_Or)) {
+ if (!Active.insert(Expr).second)
+ return false;
+ WorkList.push_back({Expr, Phase::Complete});
+ for (const MCExpr *Arg : reverse(Target->getArgs())) {
+ WorkList.push_back({Arg, Phase::CheckAbsolute});
+ WorkList.push_back({Arg});
+ }
+ continue;
+ }
+ if (const auto *Binary = dyn_cast<MCBinaryExpr>(Expr);
+ Binary && Binary->getOpcode() == MCBinaryExpr::Add) {
+ if (!Active.insert(Expr).second)
+ return false;
+ WorkList.push_back({Expr, Phase::Complete});
+ WorkList.push_back({Binary->getRHS()});
+ WorkList.push_back({Binary->getLHS()});
+ continue;
+ }
+ if (const auto *Ref = dyn_cast<MCSymbolRefExpr>(Expr)) {
+ MCSymbol &Sym = const_cast<MCSymbol &>(Ref->getSymbol());
+ if (!Ref->getKind() && Sym.isVariable() && !Sym.isWeakExternal() &&
+ !Sym.isResolving()) {
+ const MCExpr *Value = Sym.getVariableValue();
+ Sym.setIsResolving(true);
+ ResolvingSymbols.push_back(&Sym);
+ // Aliased symbols can name the same active expression. Leave the
+ // cycle diagnostic to the normal MC symbol-resolution path below.
+ if (!Active.contains(Value)) {
+ Active.insert(Expr);
+ WorkList.push_back({Expr, Phase::Complete, &Sym});
+ WorkList.push_back({Value});
+ continue;
+ }
+ }
+ }
+ if (!EvaluateLeaf(Expr))
+ return false;
}
- llvm_unreachable("Unknown expr kind!");
+
+ Res = Values.lookup(Root);
+ return Res.isAbsolute();
}
bool AMDGPUMCExpr::evaluateAsRelocatableImpl(MCValue &Res,
@@ -282,6 +421,9 @@ bool AMDGPUMCExpr::evaluateAsRelocatableImpl(MCValue &Res,
switch (Kind) {
default:
break;
+ case AGVK_Max:
+ case AGVK_Or:
+ return evaluateResourceExpr(this, Res, Asm);
case AGVK_ExtraSGPRs:
return evaluateExtraSGPRs(Res, Asm);
case AGVK_AlignTo:
diff --git a/llvm/test/CodeGen/AMDGPU/function-resource-usage.ll b/llvm/test/CodeGen/AMDGPU/function-resource-usage.ll
index a311277aa82aa..bded07ca6f2aa 100644
--- a/llvm/test/CodeGen/AMDGPU/function-resource-usage.ll
+++ b/llvm/test/CodeGen/AMDGPU/function-resource-usage.ll
@@ -495,9 +495,9 @@ define amdgpu_kernel void @usage_direct_recursion(i32 %n) #0 {
; GCN: NumVgprs: max(43, .Lmulti_stage_recurse1.num_vgpr)
; GCN: ScratchSize: 16+max(.Lmulti_stage_recurse1.private_seg_size)
; GCN-LABEL: {{^}}multi_stage_recurse1:
-; GCN: .set .Lmulti_stage_recurse1.num_vgpr, max(48, 43)
-; GCN: .set .Lmulti_stage_recurse1.num_agpr, max(0, 0)
-; GCN: .set .Lmulti_stage_recurse1.numbered_sgpr, max(34, 34)
+; GCN: .set .Lmulti_stage_recurse1.num_vgpr, 48
+; GCN: .set .Lmulti_stage_recurse1.num_agpr, 0
+; GCN: .set .Lmulti_stage_recurse1.numbered_sgpr, 34
; GCN: .set .Lmulti_stage_recurse1.private_seg_size, 16
; GCN: .set .Lmulti_stage_recurse1.uses_vcc, 1
; GCN: .set .Lmulti_stage_recurse1.uses_flat_scratch, 0
@@ -550,14 +550,14 @@ define amdgpu_kernel void @usage_multi_stage_recurse(i32 %n) #0 {
; GCN: NumVgprs: max(41, .Lmulti_stage_recurse_noattr1.num_vgpr)
; GCN: ScratchSize: 16+max(.Lmulti_stage_recurse_noattr1.private_seg_size)
; GCN-LABEL: {{^}}multi_stage_recurse_noattr1:
-; GCN: .set .Lmulti_stage_recurse_noattr1.num_vgpr, max(41, 41)
-; GCN: .set .Lmulti_stage_recurse_noattr1.num_agpr, max(0, 0)
-; GCN: .set .Lmulti_stage_recurse_noattr1.numbered_sgpr, max(57, 54)
+; GCN: .set .Lmulti_stage_recurse_noattr1.num_vgpr, 41
+; GCN: .set .Lmulti_stage_recurse_noattr1.num_agpr, 0
+; GCN: .set .Lmulti_stage_recurse_noattr1.numbered_sgpr, 57
; GCN: .set .Lmulti_stage_recurse_noattr1.private_seg_size, 16
; GCN: .set .Lmulti_stage_recurse_noattr1.uses_vcc, 1
; GCN: .set .Lmulti_stage_recurse_noattr1.uses_flat_scratch, 0
; GCN: .set .Lmulti_stage_recurse_noattr1.has_dyn_sized_stack, 0
-; GCN: .set .Lmulti_stage_recurse_noattr1.has_recursion, 0
+; GCN: .set .Lmulti_stage_recurse_noattr1.has_recursion, 1
; GCN: .set .Lmulti_stage_recurse_noattr1.has_indirect_call, 0
; GCN: TotalNumSgprs: 61
; GCN: NumVgprs: 41
diff --git a/llvm/test/CodeGen/AMDGPU/recursive-resource-usage-mcexpr.ll b/llvm/test/CodeGen/AMDGPU/recursive-resource-usage-mcexpr.ll
index b26e2784fab17..035a1c7a20242 100644
--- a/llvm/test/CodeGen/AMDGPU/recursive-resource-usage-mcexpr.ll
+++ b/llvm/test/CodeGen/AMDGPU/recursive-resource-usage-mcexpr.ll
@@ -6,7 +6,7 @@
; CHECK: .set .Lqux.num_vgpr, max(71, .Lfoo.num_vgpr)
; CHECK: .set .Lqux.num_agpr, max(0, .Lfoo.num_agpr)
; CHECK: .set .Lqux.numbered_sgpr, max(46, .Lfoo.numbered_sgpr)
-; CHECK: .set .Lqux.private_seg_size, 16
+; CHECK: .set .Lqux.private_seg_size, 16+max(.Lfoo.private_seg_size)
; CHECK: .set .Lqux.uses_vcc, or(1, .Lfoo.uses_vcc)
; CHECK: .set .Lqux.uses_flat_scratch, or(0, .Lfoo.uses_flat_scratch)
; CHECK: .set .Lqux.has_dyn_sized_stack, or(0, .Lfoo.has_dyn_sized_stack)
@@ -36,9 +36,9 @@
; CHECK: .set .Lbar.has_indirect_call, or(0, .Lbaz.has_indirect_call)
; CHECK-LABEL: {{^}}foo
-; CHECK: .set .Lfoo.num_vgpr, max(46, 71)
-; CHECK: .set .Lfoo.num_agpr, max(0, 0)
-; CHECK: .set .Lfoo.numbered_sgpr, max(71, 61)
+; CHECK: .set .Lfoo.num_vgpr, 71
+; CHECK: .set .Lfoo.num_agpr, 0
+; CHECK: .set .Lfoo.numbered_sgpr, 71
; CHECK: .set .Lfoo.private_seg_size, 16
; CHECK: .set .Lfoo.uses_vcc, 1
; CHECK: .set .Lfoo.uses_flat_scratch, 0
@@ -107,9 +107,9 @@ define amdgpu_kernel void @usefoo() {
; CHECK: .set .LD.has_indirect_call, or(0, .LC.has_indirect_call)
; CHECK-LABEL: {{^}}C
-; CHECK: .set .LC.num_vgpr, max(42, .LA.num_vgpr, 71)
-; CHECK: .set .LC.num_agpr, max(0, .LA.num_agpr, 0)
-; CHECK: .set .LC.numbered_sgpr, max(71, .LA.numbered_sgpr, 71)
+; CHECK: .set .LC.num_vgpr, max(71, .LA.num_vgpr)
+; CHECK: .set .LC.num_agpr, max(0, .LA.num_agpr)
+; CHECK: .set .LC.numbered_sgpr, max(71, .LA.numbered_sgpr)
; CHECK: .set .LC.private_seg_size, 16+max(.LA.private_seg_size)
; CHECK: .set .LC.uses_vcc, or(1, .LA.uses_vcc)
; CHECK: .set .LC.uses_flat_scratch, or(0, .LA.uses_flat_scratch)
@@ -129,9 +129,9 @@ define amdgpu_kernel void @usefoo() {
; CHECK: .set .LB.has_indirect_call, or(0, .LC.has_indirect_call)
; CHECK-LABEL: {{^}}A
-; CHECK: .set .LA.num_vgpr, max(42, 71)
-; CHECK: .set .LA.num_agpr, max(0, 0)
-; CHECK: .set .LA.numbered_sgpr, max(71, 71)
+; CHECK: .set .LA.num_vgpr, 71
+; CHECK: .set .LA.num_agpr, 0
+; CHECK: .set .LA.numbered_sgpr, 71
; CHECK: .set .LA.private_seg_size, 16
; CHECK: .set .LA.uses_vcc, 1
; CHECK: .set .LA.uses_flat_scratch, 0
diff --git a/llvm/test/CodeGen/AMDGPU/resource-usage-cycle-shared-outgoing.mir b/llvm/test/CodeGen/AMDGPU/resource-usage-cycle-shared-outgoing.mir
new file mode 100644
index 0000000000000..1a8a229fd7c5c
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/resource-usage-cycle-shared-outgoing.mir
@@ -0,0 +1,52 @@
+# RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -verify-machineinstrs \
+# RUN: -start-before=amdgpu-resource-usage %s -o - | FileCheck %s
+
+# Both cycle members also call c. When a closes the cycle, c is still
+# undefined and is reached both directly and through b. Normalize the whole
+# assignment so c occurs once, including across different callee operands.
+# CHECK-LABEL: {{^}}a:
+# CHECK: .set .La.num_vgpr, max(0, .Lc.num_vgpr){{$}}
+# CHECK: .set .La.uses_flat_scratch, or(0, .Lc.uses_flat_scratch){{$}}
+# CHECK-LABEL: {{^}}c:
+# CHECK: .set .Lc.num_vgpr, 48{{$}}
+# CHECK: .set .Lc.uses_flat_scratch, 1{{$}}
+
+--- |
+ target triple = "amdgpu9.00-amd-amdhsa"
+ define hidden void @b() norecurse { ret void }
+ define hidden void @a() norecurse { ret void }
+ define hidden void @c() norecurse { ret void }
+...
+---
+name: b
+tracksRegLiveness: true
+frameInfo:
+ hasCalls: true
+ maxCallFrameSize: 0
+body: |
+ bb.0:
+ dead $sgpr30_sgpr31 = SI_CALL undef $sgpr4_sgpr5, @a, csr_amdgpu
+ dead $sgpr30_sgpr31 = SI_CALL undef $sgpr4_sgpr5, @c, csr_amdgpu
+ S_ENDPGM 0
+...
+---
+name: a
+tracksRegLiveness: true
+frameInfo:
+ hasCalls: true
+ maxCallFrameSize: 0
+body: |
+ bb.0:
+ dead $sgpr30_sgpr31 = SI_CALL undef $sgpr4_sgpr5, @b, csr_amdgpu
+ dead $sgpr30_sgpr31 = SI_CALL undef $sgpr4_sgpr5, @c, csr_amdgpu
+ S_ENDPGM 0
+...
+---
+name: c
+tracksRegLiveness: true
+body: |
+ bb.0:
+ dead $vgpr47 = V_MOV_B32_e32 0, implicit $exec
+ dead $sgpr0_sgpr1 = S_MOV_B64 $flat_scr
+ S_ENDPGM 0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.ll b/llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.ll
new file mode 100644
index 0000000000000..5a5289d786c1a
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.ll
@@ -0,0 +1,76 @@
+; RUN: split-file %s %t
+; RUN: llc -mtriple=amdgpu9.00-amd-amdhsa %t/input.ll -o %t/sdag.s
+; RUN: FileCheck %s < %t/sdag.s
+; RUN: cat %t/sdag.s %t/check.s | llvm-mc -triple=amdgpu9.00-amd-amdhsa \
+; RUN: -filetype=obj | llvm-readobj --hex-dump=.resource_check - | \
+; RUN: FileCheck %s --check-prefix=RESOURCES
+; RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -global-isel=1 \
+; RUN: -global-isel-abort=1 %t/input.ll -o %t/gisel.s
+; RUN: FileCheck %s < %t/gisel.s
+; RUN: cat %t/gisel.s %t/check.s | llvm-mc -triple=amdgpu9.00-amd-amdhsa \
+; RUN: -filetype=obj | llvm-readobj --hex-dump=.resource_check - | \
+; RUN: FileCheck %s --check-prefix=RESOURCES
+;
+; The syntactic cycle terminates on every path: a(0) calls b(1), which
+; returns; b(0) calls a(1), which calls c(1). All norecurse attributes are
+; therefore valid. The selected emission order closes the cycle at b, and b
+; must retain c's resources through a even though it does not call c directly.
+; The outgoing c edge contributes v47 and a dynamically sized frame.
+;
+; CHECK: .set .Lc.num_vgpr, 48
+; CHECK: .amdhsa_kernel kernel_b
+; CHECK: .amdhsa_uses_dynamic_stack 1
+; CHECK: .amdhsa_next_free_vgpr 48
+; CHECK: .name: kernel_b
+; CHECK: .uses_dynamic_stack: true
+; CHECK: .vgpr_count: 48
+;
+; The first row is the final VGPR count of a, b, c, and a padding word.
+; The second row is their dynamic-stack flag. Check its own closure rather
+; than relying on the separately propagated recursion flag to hide a loss.
+; RESOURCES: Hex dump of section '.resource_check':
+; RESOURCES-NEXT: 0x00000000 30000000 30000000 30000000 00000000
+; RESOURCES-NEXT: 0x00000010 01000000 01000000 01000000 00000000
+
+;--- input.ll
+target triple = "amdgpu9.00-amd-amdhsa"
+
+define hidden void @b(i32 %mode) noinline optnone norecurse {
+ %is0 = icmp eq i32 %mode, 0
+ br i1 %is0, label %do_call, label %done
+do_call:
+ call void @a(i32 1)
+ br label %done
+done:
+ ret void
+}
+
+define hidden void @a(i32 %mode) noinline optnone norecurse {
+ %is0 = icmp eq i32 %mode, 0
+ br i1 %is0, label %call_b, label %call_c
+call_b:
+ call void @b(i32 1)
+ br label %done
+call_c:
+ call void @c(i32 1)
+ br label %done
+done:
+ ret void
+}
+
+define hidden void @c(i32 %mode) noinline optnone norecurse {
+ %frame = alloca i8, i32 %mode, align 16, addrspace(5)
+ store volatile i8 1, ptr addrspace(5) %frame, align 16
+ call void asm sideeffect "v_mov_b32 v47, 0", "~{v47}"()
+ ret void
+}
+
+define amdgpu_kernel void @kernel_b() {
+ call void @b(i32 0)
+ ret void
+}
+
+;--- check.s
+.section .resource_check, "", @progbits
+.long .La.num_vgpr, .Lb.num_vgpr, .Lc.num_vgpr, 0
+.long .La.has_dyn_sized_stack, .Lb.has_dyn_sized_stack, .Lc.has_dyn_sized_stack, 0
diff --git a/llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.mir b/llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.mir
new file mode 100644
index 0000000000000..d52fcb2b2ca6c
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.mir
@@ -0,0 +1,110 @@
+# RUN: split-file %s %t
+# RUN: cat %t/header.mir %t/a.ir %t/b.ir %t/c.ir %t/functions.mir > %t/abc.mir
+# RUN: llc -mtriple=amdgpu9.08-amd-amdhsa \
+# RUN: -start-before=amdgpu-resource-usage %t/abc.mir -o %t/abc.s
+# RUN: cat %t/abc.s %t/check.s | llvm-mc -triple=amdgpu9.08-amd-amdhsa \
+# RUN: -filetype=obj | llvm-readobj --hex-dump=.resource_check - | \
+# RUN: FileCheck %s
+# RUN: cat %t/header.mir %t/a.ir %t/c.ir %t/b.ir %t/functions.mir > %t/acb.mir
+# RUN: llc -mtriple=amdgpu9.08-amd-amdhsa \
+# RUN: -start-before=amdgpu-resource-usage %t/acb.mir -o %t/acb.s
+# RUN: cat %t/acb.s %t/check.s | llvm-mc -triple=amdgpu9.08-amd-amdhsa \
+# RUN: -filetype=obj | llvm-readobj --hex-dump=.resource_check - | \
+# RUN: FileCheck %s
+# RUN: cat %t/header.mir %t/b.ir %t/a.ir %t/c.ir %t/functions.mir > %t/bac.mir
+# RUN: llc -mtriple=amdgpu9.08-amd-amdhsa \
+# RUN: -start-before=amdgpu-resource-usage %t/bac.mir -o %t/bac.s
+# RUN: cat %t/bac.s %t/check.s | llvm-mc -triple=amdgpu9.08-amd-amdhsa \
+# RUN: -filetype=obj | llvm-readobj --hex-dump=.resource_check - | \
+# RUN: FileCheck %s
+# RUN: cat %t/header.mir %t/b.ir %t/c.ir %t/a.ir %t/functions.mir > %t/bca.mir
+# RUN: llc -mtriple=amdgpu9.08-amd-amdhsa \
+# RUN: -start-before=amdgpu-resource-usage %t/bca.mir -o %t/bca.s
+# RUN: cat %t/bca.s %t/check.s | llvm-mc -triple=amdgpu9.08-amd-amdhsa \
+# RUN: -filetype=obj | llvm-readobj --hex-dump=.resource_check - | \
+# RUN: FileCheck %s
+# RUN: cat %t/header.mir %t/c.ir %t/a.ir %t/b.ir %t/functions.mir > %t/cab.mir
+# RUN: llc -mtriple=amdgpu9.08-amd-amdhsa \
+# RUN: -start-before=amdgpu-resource-usage %t/cab.mir -o %t/cab.s
+# RUN: cat %t/cab.s %t/check.s | llvm-mc -triple=amdgpu9.08-amd-amdhsa \
+# RUN: -filetype=obj | llvm-readobj --hex-dump=.resource_check - | \
+# RUN: FileCheck %s
+# RUN: cat %t/header.mir %t/c.ir %t/b.ir %t/a.ir %t/functions.mir > %t/cba.mir
+# RUN: llc -mtriple=amdgpu9.08-amd-amdhsa \
+# RUN: -start-before=amdgpu-resource-usage %t/cba.mir -o %t/cba.s
+# RUN: cat %t/cba.s %t/check.s | llvm-mc -triple=amdgpu9.08-amd-amdhsa \
+# RUN: -filetype=obj | llvm-readobj --hex-dump=.resource_check - | \
+# RUN: FileCheck %s
+
+# Direct calls a -> {b, c}, b -> a form a call cycle with an outgoing edge to c.
+# All six emission orders must retain c's VGPR, AGPR, SGPR, flat-scratch,
+# and dynamic-stack resources at both cycle roots. In particular, flattening
+# a cycle must keep symbols that are not yet defined, and OR expressions
+# need the same closure as register maxima.
+#
+# Assemble the resource symbols into a separate section to verify their final
+# values, independently of the order-dependent printed expression spelling.
+# Every row contains the resource value for a, b, c, and a padding word.
+# CHECK: Hex dump of section '.resource_check':
+# CHECK-NEXT: 0x00000000 30000000 30000000 30000000 00000000
+# CHECK-NEXT: 0x00000010 3d000000 3d000000 3d000000 00000000
+# CHECK-NEXT: 0x00000020 47000000 47000000 47000000 00000000
+# CHECK-NEXT: 0x00000030 01000000 01000000 01000000 00000000
+# CHECK-NEXT: 0x00000040 01000000 01000000 01000000 00000000
+# CHECK-NEXT: 0x00000050 01000000 01000000 00000000 00000000
+
+#--- header.mir
+--- |
+ target triple = "amdgpu9.08-amd-amdhsa"
+#--- a.ir
+ define hidden void @a() norecurse { ret void }
+#--- b.ir
+ define hidden void @b() norecurse { ret void }
+#--- c.ir
+ define hidden void @c() norecurse { ret void }
+#--- functions.mir
+...
+---
+name: a
+tracksRegLiveness: true
+frameInfo:
+ hasCalls: true
+ maxCallFrameSize: 0
+body: |
+ bb.0:
+ dead $sgpr30_sgpr31 = SI_CALL undef $sgpr4_sgpr5, @b, csr_amdgpu
+ dead $sgpr30_sgpr31 = SI_CALL undef $sgpr4_sgpr5, @c, csr_amdgpu
+ S_ENDPGM 0
+...
+---
+name: b
+tracksRegLiveness: true
+frameInfo:
+ hasCalls: true
+ maxCallFrameSize: 0
+body: |
+ bb.0:
+ dead $sgpr30_sgpr31 = SI_CALL undef $sgpr4_sgpr5, @a, csr_amdgpu
+ S_ENDPGM 0
+...
+---
+name: c
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: variable-sized, alignment: 16 }
+body: |
+ bb.0:
+ dead $vgpr47 = V_MOV_B32_e32 0, implicit $exec
+ dead $agpr60 = V_ACCVGPR_WRITE_B32_e64 0, implicit $exec
+ dead $sgpr70 = S_MOV_B32 0
+ dead $sgpr0_sgpr1 = S_MOV_B64 $flat_scr
+ S_ENDPGM 0
+...
+#--- check.s
+.section .resource_check, "", @progbits
+.long .La.num_vgpr, .Lb.num_vgpr, .Lc.num_vgpr, 0
+.long .La.num_agpr, .Lb.num_agpr, .Lc.num_agpr, 0
+.long .La.numbered_sgpr, .Lb.numbered_sgpr, .Lc.numbered_sgpr, 0
+.long .La.uses_flat_scratch, .Lb.uses_flat_scratch, .Lc.uses_flat_scratch, 0
+.long .La.has_dyn_sized_stack, .Lb.has_dyn_sized_stack, .Lc.has_dyn_sized_stack, 0
+.long .La.has_recursion, .Lb.has_recursion, .Lc.has_recursion, 0
diff --git a/llvm/test/CodeGen/AMDGPU/resource-usage-norecurse-cycle.ll b/llvm/test/CodeGen/AMDGPU/resource-usage-norecurse-cycle.ll
new file mode 100644
index 0000000000000..59d5b66bb3b71
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/resource-usage-norecurse-cycle.ll
@@ -0,0 +1,112 @@
+; RUN: split-file %s %t
+; RUN: llc -O0 -mtriple=amdgpu9.00-amd-amdhsa \
+; RUN: < %t/ab.ll | FileCheck %s --check-prefix=AB
+; RUN: llc -O0 -mtriple=amdgpu9.00-amd-amdhsa \
+; RUN: < %t/ba.ll | FileCheck %s --check-prefix=BA
+; RUN: llc -O0 -mtriple=amdgpu9.00-amd-amdhsa --amdhsa-code-object-version=4 \
+; RUN: -filetype=obj %t/ab.ll -o %t/v4.o
+; RUN: llvm-readobj --hex-dump=.rodata %t/v4.o | FileCheck %s --check-prefix=V4
+; RUN: llc -O0 -mtriple=amdgpu9.00-amd-amdhsa --amdhsa-code-object-version=4 \
+; RUN: %t/ab.ll -o %t/v4.s
+; RUN: llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/v4.s -o %t/v4-roundtrip.o
+; RUN: llvm-readobj --hex-dump=.rodata %t/v4-roundtrip.o | FileCheck %s --check-prefix=V4
+; RUN: llc -O0 -mtriple=amdgpu9.00-amd-amdhsa --amdhsa-code-object-version=5 \
+; RUN: -filetype=obj %t/ab.ll -o %t/v5.o
+; RUN: llvm-readobj --hex-dump=.rodata %t/v5.o | FileCheck %s --check-prefix=V5
+; RUN: llc -O0 -mtriple=amdgpu9.00-amd-amdhsa --amdhsa-code-object-version=5 \
+; RUN: %t/ab.ll -o %t/v5.s
+; RUN: llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/v5.s -o %t/v5-roundtrip.o
+; RUN: llvm-readobj --hex-dump=.rodata %t/v5-roundtrip.o | FileCheck %s --check-prefix=V5
+
+; Both entry paths are finite and satisfy norecurse. The private-segment
+; expression drops a syntactic cycle edge, so V5 and later must request
+; dynamic-stack provisioning at both roots, regardless of emission order.
+; AB-COUNT-2: .amdhsa_uses_dynamic_stack 1
+; AB-NOT: .amdhsa_uses_dynamic_stack 0
+; BA-COUNT-2: .amdhsa_uses_dynamic_stack 1
+; BA-NOT: .amdhsa_uses_dynamic_stack 0
+
+; Each kernel descriptor is 64 bytes. Its 16-bit properties field at offset 56
+; contains the dynamic-stack bit (0x0800) only in V5 and later. Check the raw
+; bytes because the V4 assembly printer omits the directive even if the direct
+; object path incorrectly sets the bit. Both emission paths must agree.
+; V4: Hex dump of section '.rodata':
+; V4: 0x00000030 {{[0-9a-f]+}} {{[0-9a-f]+}} 3f000000
+; V4: 0x00000070 {{[0-9a-f]+}} {{[0-9a-f]+}} 3f000000
+; V5: Hex dump of section '.rodata':
+; V5: 0x00000030 {{[0-9a-f]+}} {{[0-9a-f]+}} 3f080000
+; V5: 0x00000070 {{[0-9a-f]+}} {{[0-9a-f]+}} 3f080000
+
+;--- ab.ll
+target triple = "amdgpu9.00-amd-amdhsa"
+
+define hidden void @a(i1 %go) noinline norecurse {
+entry:
+ %x = alloca [32 x i8], align 4, addrspace(5)
+ store volatile i8 1, ptr addrspace(5) %x
+ br i1 %go, label %call, label %ret
+call:
+ call void @b(i1 false)
+ br label %ret
+ret:
+ ret void
+}
+
+define hidden void @b(i1 %go) noinline norecurse {
+entry:
+ %x = alloca [64 x i8], align 4, addrspace(5)
+ store volatile i8 1, ptr addrspace(5) %x
+ br i1 %go, label %call, label %ret
+call:
+ call void @a(i1 false)
+ br label %ret
+ret:
+ ret void
+}
+
+define amdgpu_kernel void @k_a() {
+ call void @a(i1 true)
+ ret void
+}
+
+define amdgpu_kernel void @k_b() {
+ call void @b(i1 true)
+ ret void
+}
+
+;--- ba.ll
+target triple = "amdgpu9.00-amd-amdhsa"
+
+define hidden void @b(i1 %go) noinline norecurse {
+entry:
+ %x = alloca [64 x i8], align 4, addrspace(5)
+ store volatile i8 1, ptr addrspace(5) %x
+ br i1 %go, label %call, label %ret
+call:
+ call void @a(i1 false)
+ br label %ret
+ret:
+ ret void
+}
+
+define hidden void @a(i1 %go) noinline norecurse {
+entry:
+ %x = alloca [32 x i8], align 4, addrspace(5)
+ store volatile i8 1, ptr addrspace(5) %x
+ br i1 %go, label %call, label %ret
+call:
+ call void @b(i1 false)
+ br label %ret
+ret:
+ ret void
+}
+
+define amdgpu_kernel void @k_a() {
+ call void @a(i1 true)
+ ret void
+}
+
+define amdgpu_kernel void @k_b() {
+ call void @b(i1 true)
+ ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/resource-usage-private-stack-dag.test b/llvm/test/CodeGen/AMDGPU/resource-usage-private-stack-dag.test
new file mode 100644
index 0000000000000..194f4e003d14c
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/resource-usage-private-stack-dag.test
@@ -0,0 +1,61 @@
+# RUN: %python %s > %t.mir
+# RUN: llc -mtriple=amdgpu9.00-amd-amdhsa \
+# RUN: -start-before=amdgpu-resource-usage %t.mir -o %t.s
+# RUN: FileCheck %s --check-prefix=ASM < %t.s
+# RUN: llc -mtriple=amdgpu9.00-amd-amdhsa \
+# RUN: -start-before=amdgpu-resource-usage -filetype=obj %t.mir -o %t.o
+# RUN: llvm-readobj --notes %t.o | FileCheck %s --check-prefix=META
+
+# This acyclic graph has 64 functions, each with a 16-byte final frame.
+# Every function calls the next two functions, so its private-size expression
+# shares subexpressions along exponentially many paths. The longest call chain
+# visits all 64 functions and requires exactly 1024 bytes. Exercise real kernel
+# descriptor and metadata emission through both the assembly and object paths.
+# ASM-LABEL: .amdhsa_kernel kernel
+# ASM: .amdhsa_private_segment_fixed_size 1024
+# ASM: .amdhsa_uses_dynamic_stack 0
+# META: .name: kernel
+# META: .private_segment_fixed_size: 1024
+# META: .uses_dynamic_stack: false
+
+COUNT = 64
+
+print('--- |')
+print(' target triple = "amdgpu9.00-amd-amdhsa"')
+for i in range(COUNT):
+ print(f' define hidden void @f{i}() norecurse {{ ret void }}')
+print(' define amdgpu_kernel void @kernel() norecurse { ret void }')
+print('...')
+
+for i in range(COUNT):
+ callees = range(i + 1, min(i + 3, COUNT))
+ print('---')
+ print(f'name: f{i}')
+ print('tracksRegLiveness: true')
+ print('frameInfo:')
+ print(' stackSize: 16')
+ if callees:
+ print(' hasCalls: true')
+ print(' maxCallFrameSize: 0')
+ print('body: |')
+ print(' bb.0:')
+ for callee in callees:
+ print(' dead $sgpr30_sgpr31 = SI_CALL '
+ f'undef $sgpr4_sgpr5, @f{callee}, csr_amdgpu')
+ print(' S_ENDPGM 0')
+ print('...')
+
+print('''---
+name: kernel
+tracksRegLiveness: true
+machineFunctionInfo:
+ isEntryFunction: true
+frameInfo:
+ hasCalls: true
+ maxCallFrameSize: 0
+body: |
+ bb.0:
+ dead $sgpr30_sgpr31 = SI_CALL undef $sgpr4_sgpr5, @f0, csr_amdgpu
+ S_ENDPGM 0
+...
+''')
diff --git a/llvm/test/MC/AMDGPU/idempotent-expr-errors.s b/llvm/test/MC/AMDGPU/idempotent-expr-errors.s
new file mode 100644
index 0000000000000..4c4ee2c7042c4
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/idempotent-expr-errors.s
@@ -0,0 +1,101 @@
+// RUN: split-file %s %t
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/unresolved.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=UNRESOLVED
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/relocatable.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=RELOCATABLE
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/relocatable-alias.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=RELOCATABLE --implicit-check-not="cyclic dependency"
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/max-unresolved-first.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=EARLY-FAILURE --implicit-check-not="cyclic dependency"
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/max-relocatable-first.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=EARLY-FAILURE --implicit-check-not="cyclic dependency"
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/or-unresolved-first.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=EARLY-FAILURE --implicit-check-not="cyclic dependency"
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/or-relocatable-first.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=EARLY-FAILURE --implicit-check-not="cyclic dependency"
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/max-cycle.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=CYCLE
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/or-cycle.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=CYCLE
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/alias-cycle.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=CYCLE
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/mixed-cycle.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=CYCLE
+// RUN: not llvm-mc -triple=amdgpu9.00-amd-amdhsa -filetype=obj %t/binary-cycle.s -o /dev/null 2>&1 | FileCheck %s --check-prefix=CYCLE
+
+// A known all-ones OR operand does not make an unresolved or relocatable
+// operand absolute. Iterative evaluation must preserve the existing rejection.
+// UNRESOLVED: error: expected relocatable expression
+// RELOCATABLE: error: expected relocatable expression
+
+// Evaluate operands in order and stop on the first nonabsolute operand.
+// A later cycle must not be visited after the target expression has failed.
+// EARLY-FAILURE: error: expected relocatable expression
+
+// A completed shared expression is reusable; an active symbol cycle is not.
+// CYCLE: error: cyclic dependency detected for symbol
+
+//--- unresolved.s
+.data
+.long or(-1, missing)
+
+//--- relocatable.s
+.data
+label:
+.byte 0
+.long or(-1, label)
+
+//--- max-cycle.s
+.set a, max(b, 1)
+.set b, max(a, 2)
+.data
+.long a
+
+//--- or-cycle.s
+.set a, or(b, 1)
+.set b, or(a, 2)
+.data
+.long a
+
+//--- alias-cycle.s
+.set a, alias
+.set alias, b
+.set b, max(a, 2)
+.data
+.long a
+
+//--- mixed-cycle.s
+.set a, max(b, 1)
+.set b, or(a, 2)
+.data
+.long a
+
+//--- binary-cycle.s
+.set a, max(b, 1)
+.set b, a + 1
+.data
+.long a
+
+//--- relocatable-alias.s
+.data
+label:
+.byte 0
+.set alias, label + 1
+.long or(-1, alias)
+
+//--- max-unresolved-first.s
+.data
+.long max(missing, .Lcycle_a)
+.set .Lcycle_a, max(.Lcycle_b, 1)
+.set .Lcycle_b, max(.Lcycle_a, 2)
+
+//--- max-relocatable-first.s
+.data
+label:
+.byte 0
+.long max(label, .Lcycle_a)
+.set .Lcycle_a, max(.Lcycle_b, 1)
+.set .Lcycle_b, max(.Lcycle_a, 2)
+
+//--- or-unresolved-first.s
+.data
+.long or(missing, .Lcycle_a)
+.set .Lcycle_a, max(.Lcycle_b, 1)
+.set .Lcycle_b, max(.Lcycle_a, 2)
+
+//--- or-relocatable-first.s
+.data
+label:
+.byte 0
+.long or(label, .Lcycle_a)
+.set .Lcycle_a, max(.Lcycle_b, 1)
+.set .Lcycle_b, max(.Lcycle_a, 2)
diff --git a/llvm/test/MC/AMDGPU/idempotent-expr-evaluation.s b/llvm/test/MC/AMDGPU/idempotent-expr-evaluation.s
new file mode 100644
index 0000000000000..1449b720447dc
--- /dev/null
+++ b/llvm/test/MC/AMDGPU/idempotent-expr-evaluation.s
@@ -0,0 +1,163 @@
+// RUN: llvm-mc -triple=amdgpu9.08-amd-amdhsa -filetype=obj %s -o %t
+// RUN: llvm-readobj --hex-dump=.data %t | FileCheck %s
+
+// Both forward DAGs contain only 33 nodes but have over four billion paths.
+// Evaluate shared subexpressions once, with a fresh state for every query.
+// All maximum operands are negative; OR operands are arbitrary bitsets.
+// CHECK: Hex dump of section '.data':
+// CHECK-NEXT: 0x00000000 f9ffffff ffff0000 feffffff 0a000000
+// CHECK-NEXT: 0x00000010 3a000000 fdffffff 40210000 ffffffff
+// CHECK-NEXT: 0x00000020 f7ffffff ffffffff 00000000 01000080
+// CHECK-NEXT: 0x00000030 02000000 07000000 00000000 00000000
+// CHECK-NEXT: 0x00000040 02000000 02000000 07000000 00000000
+// CHECK-NEXT: 0x00000050 21000000 00000000 fcffffff ffffffff
+
+.set .Lmax0, max(.Lmax1, .Lmax1, -9)
+.set .Lmax1, max(.Lmax2, .Lmax2, -9)
+.set .Lmax2, max(.Lmax3, .Lmax3, -9)
+.set .Lmax3, max(.Lmax4, .Lmax4, -9)
+.set .Lmax4, max(.Lmax5, .Lmax5, -9)
+.set .Lmax5, max(.Lmax6, .Lmax6, -9)
+.set .Lmax6, max(.Lmax7, .Lmax7, -9)
+.set .Lmax7, max(.Lmax8, .Lmax8, -9)
+.set .Lmax8, max(.Lmax9, .Lmax9, -9)
+.set .Lmax9, max(.Lmax10, .Lmax10, -9)
+.set .Lmax10, max(.Lmax11, .Lmax11, -9)
+.set .Lmax11, max(.Lmax12, .Lmax12, -9)
+.set .Lmax12, max(.Lmax13, .Lmax13, -9)
+.set .Lmax13, max(.Lmax14, .Lmax14, -9)
+.set .Lmax14, max(.Lmax15, .Lmax15, -9)
+.set .Lmax15, max(.Lmax16, .Lmax16, -9)
+.set .Lmax16, max(.Lmax17, .Lmax17, -9)
+.set .Lmax17, max(.Lmax18, .Lmax18, -9)
+.set .Lmax18, max(.Lmax19, .Lmax19, -9)
+.set .Lmax19, max(.Lmax20, .Lmax20, -9)
+.set .Lmax20, max(.Lmax21, .Lmax21, -9)
+.set .Lmax21, max(.Lmax22, .Lmax22, -9)
+.set .Lmax22, max(.Lmax23, .Lmax23, -9)
+.set .Lmax23, max(.Lmax24, .Lmax24, -9)
+.set .Lmax24, max(.Lmax25, .Lmax25, -9)
+.set .Lmax25, max(.Lmax26, .Lmax26, -9)
+.set .Lmax26, max(.Lmax27, .Lmax27, -9)
+.set .Lmax27, max(.Lmax28, .Lmax28, -9)
+.set .Lmax28, max(.Lmax29, .Lmax29, -9)
+.set .Lmax29, max(.Lmax30, .Lmax30, -9)
+.set .Lmax30, max(.Lmax31, .Lmax31, -9)
+.set .Lmax31, max(.Lmax32, .Lmax32, -9)
+.set .Lmax32, -7
+
+.set .Lor0, or(.Lor1, .Lor1, 1)
+.set .Lor1, or(.Lor2, .Lor2, 2)
+.set .Lor2, or(.Lor3, .Lor3, 4)
+.set .Lor3, or(.Lor4, .Lor4, 8)
+.set .Lor4, or(.Lor5, .Lor5, 16)
+.set .Lor5, or(.Lor6, .Lor6, 32)
+.set .Lor6, or(.Lor7, .Lor7, 64)
+.set .Lor7, or(.Lor8, .Lor8, 128)
+.set .Lor8, or(.Lor9, .Lor9, 256)
+.set .Lor9, or(.Lor10, .Lor10, 512)
+.set .Lor10, or(.Lor11, .Lor11, 1024)
+.set .Lor11, or(.Lor12, .Lor12, 2048)
+.set .Lor12, or(.Lor13, .Lor13, 4096)
+.set .Lor13, or(.Lor14, .Lor14, 8192)
+.set .Lor14, or(.Lor15, .Lor15, 16384)
+.set .Lor15, or(.Lor16, .Lor16, 32768)
+.set .Lor16, or(.Lor17, .Lor17, 1)
+.set .Lor17, or(.Lor18, .Lor18, 2)
+.set .Lor18, or(.Lor19, .Lor19, 4)
+.set .Lor19, or(.Lor20, .Lor20, 8)
+.set .Lor20, or(.Lor21, .Lor21, 16)
+.set .Lor21, or(.Lor22, .Lor22, 32)
+.set .Lor22, or(.Lor23, .Lor23, 64)
+.set .Lor23, or(.Lor24, .Lor24, 128)
+.set .Lor24, or(.Lor25, .Lor25, 256)
+.set .Lor25, or(.Lor26, .Lor26, 512)
+.set .Lor26, or(.Lor27, .Lor27, 1024)
+.set .Lor27, or(.Lor28, .Lor28, 2048)
+.set .Lor28, or(.Lor29, .Lor29, 4096)
+.set .Lor29, or(.Lor30, .Lor30, 8192)
+.set .Lor30, or(.Lor31, .Lor31, 16384)
+.set .Lor31, or(.Lor32, .Lor32, 32768)
+.set .Lor32, 0
+
+// Multiple aliases share the same expression. A completed shared value is
+// distinct from a value that is still active in the current traversal.
+.set .Lalias_a, .Lshared
+.set .Lalias_b, .Lshared
+.set .Lshared, max(.Llate_alias, -9)
+.set .Laliases, max(.Lalias_a, .Lalias_b, .Lshared)
+.set .Llate_alias, -2
+
+// Preserve operation boundaries and the ordinary evaluation of binary leaves.
+.set .Lmixed_max, max(or(.Lbits, 4), max(.Lmax_leaf, 3), .Lbinary + 2)
+.set .Lbits, 2
+.set .Lmax_leaf, 5
+.set .Lbinary, 8
+.set .Lmixed_or, or(max(.La, 4), or(.Lb, 2), .Lc + 1)
+.set .La, 8
+.set .Lb, 16
+.set .Lc, 31
+
+.data
+.long .Lmax0, .Lor0, .Laliases, .Lmixed_max
+.long .Lmixed_or, max(-9, -3, -2147483648), or(0x100, 0x40, 0x2000), max(-1)
+.quad max(-9, -9223372036854775808)
+.quad or(0x8000000000000000, 0x100000000)
+
+// A later evaluation must observe a reassigned symbol's new value.
+.set .Lmutable, 1
+.long max(.Lmutable, 2)
+.set .Lmutable, 7
+.long max(.Lmutable, 2)
+.long 0, 0
+
+// An earlier expression retains its symbol version when the source spelling
+// is reassigned. Following an alias must not switch to the latest named value.
+.set .Ldelayed, max(.Lfuture, 2)
+.set .Lfuture, 1
+.long .Ldelayed
+.set .Lfuture, 7
+.long .Ldelayed
+.long max(.Lfuture, 2), 0
+
+// Private-stack expressions interleave addition and shared maxima. A separate
+// state for each maximum would re-expand the same additive DAG exponentially.
+.set .Ladd0, 1 + max(.Ladd1, .Ladd1)
+.set .Ladd1, 1 + max(.Ladd2, .Ladd2)
+.set .Ladd2, 1 + max(.Ladd3, .Ladd3)
+.set .Ladd3, 1 + max(.Ladd4, .Ladd4)
+.set .Ladd4, 1 + max(.Ladd5, .Ladd5)
+.set .Ladd5, 1 + max(.Ladd6, .Ladd6)
+.set .Ladd6, 1 + max(.Ladd7, .Ladd7)
+.set .Ladd7, 1 + max(.Ladd8, .Ladd8)
+.set .Ladd8, 1 + max(.Ladd9, .Ladd9)
+.set .Ladd9, 1 + max(.Ladd10, .Ladd10)
+.set .Ladd10, 1 + max(.Ladd11, .Ladd11)
+.set .Ladd11, 1 + max(.Ladd12, .Ladd12)
+.set .Ladd12, 1 + max(.Ladd13, .Ladd13)
+.set .Ladd13, 1 + max(.Ladd14, .Ladd14)
+.set .Ladd14, 1 + max(.Ladd15, .Ladd15)
+.set .Ladd15, 1 + max(.Ladd16, .Ladd16)
+.set .Ladd16, 1 + max(.Ladd17, .Ladd17)
+.set .Ladd17, 1 + max(.Ladd18, .Ladd18)
+.set .Ladd18, 1 + max(.Ladd19, .Ladd19)
+.set .Ladd19, 1 + max(.Ladd20, .Ladd20)
+.set .Ladd20, 1 + max(.Ladd21, .Ladd21)
+.set .Ladd21, 1 + max(.Ladd22, .Ladd22)
+.set .Ladd22, 1 + max(.Ladd23, .Ladd23)
+.set .Ladd23, 1 + max(.Ladd24, .Ladd24)
+.set .Ladd24, 1 + max(.Ladd25, .Ladd25)
+.set .Ladd25, 1 + max(.Ladd26, .Ladd26)
+.set .Ladd26, 1 + max(.Ladd27, .Ladd27)
+.set .Ladd27, 1 + max(.Ladd28, .Ladd28)
+.set .Ladd28, 1 + max(.Ladd29, .Ladd29)
+.set .Ladd29, 1 + max(.Ladd30, .Ladd30)
+.set .Ladd30, 1 + max(.Ladd31, .Ladd31)
+.set .Ladd31, 1 + max(.Ladd32, .Ladd32)
+.set .Ladd32, 1
+.quad .Ladd0
+
+// The cached absolute-addition path must preserve MC's 64-bit wrapping.
+.set .Lwrapped, max(.Lsigned_max + 1, -4)
+.set .Lsigned_max, 0x7fffffffffffffff
+.quad .Lwrapped
>From 222715d90ed9b1ba3ae659561b9218505de82e15 Mon Sep 17 00:00:00 2001
From: mmiftahx <mmiftah.duna at gmail.com>
Date: Sun, 13 Sep 2026 23:58:32 -0500
Subject: [PATCH 2/2] fixup! [AMDGPU] Preserve resource requirements through
call-graph cycles
---
.../Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp | 15 +++++++++------
.../resource-usage-cycle-shared-outgoing.mir | 3 +--
.../AMDGPU/resource-usage-direct-call-cycle.mir | 1 -
3 files changed, 10 insertions(+), 9 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
index 86459a70063e8..14987588f4f1f 100644
--- a/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
+++ b/llvm/lib/Target/AMDGPU/MCTargetDesc/AMDGPUMCExpr.cpp
@@ -300,13 +300,13 @@ static bool evaluateResourceExpr(const AMDGPUMCExpr *Root, MCValue &Res,
Phase Step = Phase::Visit;
MCSymbol *ResolvingSymbol = nullptr;
};
- SmallVector<WorkItem, 16> WorkList{{Root}};
+ SmallVector<WorkItem, 16> WorkList = {{Root}};
DenseMap<const MCExpr *, MCValue> Values;
SmallPtrSet<const MCExpr *, 16> Active;
SmallVector<MCSymbol *, 8> ResolvingSymbols;
// Match MCExpr's resolution guard, including on failure. Cache values only
// for this query: symbol definitions and assembler layout can change later.
- auto ClearResolving = scope_exit([&] {
+ scope_exit ClearResolving([&] {
for (MCSymbol *Sym : ResolvingSymbols)
Sym->setIsResolving(false);
});
@@ -333,14 +333,16 @@ static bool evaluateResourceExpr(const AMDGPUMCExpr *Root, MCValue &Res,
if (MCSymbol *Sym = Item.ResolvingSymbol) {
MCValue Value = Values.lookup(Sym->getVariableValue());
Sym->setIsResolving(false);
- if (Value.isAbsolute())
+ if (Value.isAbsolute()) {
Values.try_emplace(Expr, Value);
- // The normal resolver preserves the identity of relocatable aliases.
- // Clear our guard first so it does not diagnose a spurious cycle.
- else if (!EvaluateLeaf(Expr))
+ } else if (!EvaluateLeaf(Expr)) {
+ // The normal resolver preserves the identity of relocatable aliases.
+ // Clear our guard first so it does not diagnose a spurious cycle.
return false;
+ }
continue;
}
+
if (const auto *Binary = dyn_cast<MCBinaryExpr>(Expr)) {
MCValue LHS = Values.lookup(Binary->getLHS());
MCValue RHS = Values.lookup(Binary->getRHS());
@@ -390,6 +392,7 @@ static bool evaluateResourceExpr(const AMDGPUMCExpr *Root, MCValue &Res,
WorkList.push_back({Binary->getLHS()});
continue;
}
+
if (const auto *Ref = dyn_cast<MCSymbolRefExpr>(Expr)) {
MCSymbol &Sym = const_cast<MCSymbol &>(Ref->getSymbol());
if (!Ref->getKind() && Sym.isVariable() && !Sym.isWeakExternal() &&
diff --git a/llvm/test/CodeGen/AMDGPU/resource-usage-cycle-shared-outgoing.mir b/llvm/test/CodeGen/AMDGPU/resource-usage-cycle-shared-outgoing.mir
index 1a8a229fd7c5c..6491457785aca 100644
--- a/llvm/test/CodeGen/AMDGPU/resource-usage-cycle-shared-outgoing.mir
+++ b/llvm/test/CodeGen/AMDGPU/resource-usage-cycle-shared-outgoing.mir
@@ -1,4 +1,4 @@
-# RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -verify-machineinstrs \
+# RUN: llc -mtriple=amdgpu9.00-amd-amdhsa \
# RUN: -start-before=amdgpu-resource-usage %s -o - | FileCheck %s
# Both cycle members also call c. When a closes the cycle, c is still
@@ -12,7 +12,6 @@
# CHECK: .set .Lc.uses_flat_scratch, 1{{$}}
--- |
- target triple = "amdgpu9.00-amd-amdhsa"
define hidden void @b() norecurse { ret void }
define hidden void @a() norecurse { ret void }
define hidden void @c() norecurse { ret void }
diff --git a/llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.mir b/llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.mir
index d52fcb2b2ca6c..30e8b55881530 100644
--- a/llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.mir
+++ b/llvm/test/CodeGen/AMDGPU/resource-usage-direct-call-cycle.mir
@@ -55,7 +55,6 @@
#--- header.mir
--- |
- target triple = "amdgpu9.08-amd-amdhsa"
#--- a.ir
define hidden void @a() norecurse { ret void }
#--- b.ir
More information about the llvm-commits
mailing list