[llvm] [AMDGPU] Replace amdgpu-num-* attributes with stress options (PR #213924)
via llvm-commits
llvm-commits at lists.llvm.org
Tue Aug 4 05:32:31 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-backend-amdgpu
Author: Romanov Vlad (romanovvlad)
<details>
<summary>Changes</summary>
In https://github.com/llvm/llvm-project/pull/210907, `-amdgpu-stress-*` options were added to `SIRegisterInfo::getReservedRegs`. However, several components that check register availability bypass reserved registers entirely — for example, GCNNSAReassign calls `getMaxNumVGPRs` or `getAddressableNumArchVGPRs` directly, so the stress options had no effect on them.
This patch replaces amdgpu-num-* attribute handling with options and refactors the `getMaxNum*` function hierarchy.
Before:
```
getMaxNumVGPRs(WavesPerEU) [hardware limit for given occupancy]
|
v
getBaseMaxNumVGPRs(F) [reads amdgpu-num-vgpr attribute, clamps]
|
v
getMaxNumVGPRs(F) [returns total budget as single number]
|
+---> getMaxNumVGPRs(MF) [forwards to F]
|
v
getMaxNumVectorRegs(F) [calls getMaxNumVGPRs(F), then splits
| result into (VGPRs, AGPRs) pair]
v
getReservedRegs(MF) [reserves phys regs above the limit;
| stress options applied HERE separately]
v
getNumAllocatableRegs(VGPR_32) [total physical - reserved; used by
scheduler ExcessLimit -- but NOT used by
RewriteMFMAFormStage or NSA Reassign]
```
The problem: stress options were applied as a separate override inside getReservedRegs, on top of the values from getMaxNumVectorRegs. Anything calling getMaxNumVGPRs or getMaxNumVectorRegs directly never saw them.
After:
```
getMaxNumVGPRs(WavesPerEU) [hardware limit for given occupancy]
|
v
getMaxNumVectorRegs(F) [PRIMARY: computes occupancy budget,
| splits into (VGPRs, AGPRs) pair,
| applies stress overrides]
|
+---> getMaxNumVGPRs(F) [wrapper: VGPRs + AGPRs on gfx90a,
| | VGPRs only on gfx908/other]
| v
| getMaxNumVGPRs(MF) [forwards to F]
|
+---> getReservedRegs(MF) [reserves phys regs above the limit]
|
v
getNumAllocatableRegs [stress-aware via the chain above;
now also used by RewriteMFMAFormStage]
```
Now stress overrides are applied inside getMaxNumVectorRegs itself, so every downstream consumer sees them - both the getMaxNumVGPRs chain and the getReservedRegs -> getNumAllocatableRegs chain.
Additional changes:
- Some callers switched from getMaxNumSGPRs(Function) to getMaxNumSGPRs(MachineFunction) to use the actual reserved SGPR count (based on hasFlatScratchInit()) rather than the conservative subtarget-level estimate (hasFlatAddressSpace()). The Function overload is removed as it has no remaining callers.
- RewriteMFMAFormStage now uses getNumAllocatableRegs(VGPR_32) instead of getAddressableNumArchVGPRs(), which always returned the hardware constant (256) and was never stress-aware.
- Tests that used amdgpu-num-vgpr/amdgpu-num-sgpr attributes are updated to use the corresponding stress options. Tests with multiple functions using different per-function limits are split into separate files, since the stress options are global.
---
Patch is 444.00 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/213924.diff
55 Files Affected:
- (modified) llvm/lib/Target/AMDGPU/GCNRegPressure.cpp (+1-2)
- (modified) llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp (+4-5)
- (modified) llvm/lib/Target/AMDGPU/GCNSubtarget.cpp (+32-61)
- (modified) llvm/lib/Target/AMDGPU/GCNSubtarget.h (+6-44)
- (modified) llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp (-20)
- (modified) llvm/test/CodeGen/AMDGPU/agpr-remat.ll (+2-2)
- (removed) llvm/test/CodeGen/AMDGPU/attr-amdgpu-num-sgpr.ll (-131)
- (modified) llvm/test/CodeGen/AMDGPU/attr-unparseable.ll (-21)
- (added) llvm/test/CodeGen/AMDGPU/cc-update-stress.ll (+89)
- (modified) llvm/test/CodeGen/AMDGPU/cc-update.ll (-83)
- (modified) llvm/test/CodeGen/AMDGPU/frame-lowering-entry-all-sgpr-used.mir (+2-2)
- (added) llvm/test/CodeGen/AMDGPU/gfx-callable-return-types-stress.ll (+1169)
- (modified) llvm/test/CodeGen/AMDGPU/gfx-callable-return-types.ll (-1166)
- (added) llvm/test/CodeGen/AMDGPU/hsa-metadata-kernel-code-props-sgpr-spill.ll (+56)
- (renamed) llvm/test/CodeGen/AMDGPU/hsa-metadata-kernel-code-props-vgpr-spill.ll (+11-6)
- (modified) llvm/test/CodeGen/AMDGPU/hsa-metadata-kernel-code-props.ll (+4-126)
- (modified) llvm/test/CodeGen/AMDGPU/machine-scheduler-sink-trivial-remats-attr.mir (-420)
- (added) llvm/test/CodeGen/AMDGPU/machine-scheduler-sink-trivial-remats-sgpr-stress.mir (+330)
- (added) llvm/test/CodeGen/AMDGPU/machine-scheduler-sink-trivial-remats-vgpr-stress.mir (+101)
- (added) llvm/test/CodeGen/AMDGPU/nsa-reassign-stress.ll (+24)
- (modified) llvm/test/CodeGen/AMDGPU/nsa-reassign.ll (-22)
- (modified) llvm/test/CodeGen/AMDGPU/nsa-reassign.mir (+3-5)
- (modified) llvm/test/CodeGen/AMDGPU/partial-regcopy-and-spill-missed-at-regalloc.ll (+5-5)
- (added) llvm/test/CodeGen/AMDGPU/pei-build-spill-partial-agpr-vgpr1.mir (+93)
- (added) llvm/test/CodeGen/AMDGPU/pei-build-spill-partial-agpr-vgpr3.mir (+109)
- (added) llvm/test/CodeGen/AMDGPU/pei-build-spill-partial-agpr-vgpr4.mir (+68)
- (added) llvm/test/CodeGen/AMDGPU/pei-build-spill-partial-agpr-vgpr5.mir (+147)
- (removed) llvm/test/CodeGen/AMDGPU/pei-build-spill-partial-agpr.mir (-399)
- (modified) llvm/test/CodeGen/AMDGPU/preserve-wwm-copy-dst-reg.ll (+3-3)
- (added) llvm/test/CodeGen/AMDGPU/schedule-amdgpu-trackers-stress.ll (+38)
- (modified) llvm/test/CodeGen/AMDGPU/schedule-amdgpu-trackers.ll (-34)
- (modified) llvm/test/CodeGen/AMDGPU/schedule-regpressure-limit-clustering.ll (+2-3)
- (modified) llvm/test/CodeGen/AMDGPU/si-lower-sgpr-spills-cycle-header.ll (+2-2)
- (added) llvm/test/CodeGen/AMDGPU/spill-agpr-vgpr10.ll (+81)
- (added) llvm/test/CodeGen/AMDGPU/spill-agpr-vgpr12.ll (+135)
- (added) llvm/test/CodeGen/AMDGPU/spill-agpr-vgpr32.ll (+186)
- (added) llvm/test/CodeGen/AMDGPU/spill-agpr-vgpr6.ll (+130)
- (removed) llvm/test/CodeGen/AMDGPU/spill-agpr.ll (-493)
- (added) llvm/test/CodeGen/AMDGPU/spill-offset-calculation-sgpr16.ll (+90)
- (added) llvm/test/CodeGen/AMDGPU/spill-offset-calculation-sgpr18.ll (+91)
- (modified) llvm/test/CodeGen/AMDGPU/spill-offset-calculation.ll (-166)
- (modified) llvm/test/CodeGen/AMDGPU/spill-sgpr-stack-no-sgpr.ll (+2-2)
- (modified) llvm/test/CodeGen/AMDGPU/spill-vector-superclass.ll (+2-2)
- (added) llvm/test/CodeGen/AMDGPU/spill-vgpr-stress10.ll (+27)
- (added) llvm/test/CodeGen/AMDGPU/spill-vgpr-stress11.ll (+56)
- (modified) llvm/test/CodeGen/AMDGPU/spill-vgpr-to-agpr.ll (+2-2)
- (modified) llvm/test/CodeGen/AMDGPU/spill-vgpr.ll (-74)
- (modified) llvm/test/CodeGen/AMDGPU/splitkit.mir (+4-6)
- (modified) llvm/test/CodeGen/AMDGPU/undefined-physreg-sgpr-spill.mir (+2-2)
- (added) llvm/test/CodeGen/AMDGPU/vgpr-mark-last-scratch-load-stress11.ll (+214)
- (added) llvm/test/CodeGen/AMDGPU/vgpr-mark-last-scratch-load-stress6.ll (+69)
- (modified) llvm/test/CodeGen/AMDGPU/vgpr-mark-last-scratch-load.ll (-280)
- (modified) llvm/test/CodeGen/AMDGPU/vgpr-mark-last-scratch-load.mir (+5-5)
- (modified) llvm/test/CodeGen/AMDGPU/whole-wave-register-copy.ll (+2-2)
- (modified) llvm/test/CodeGen/AMDGPU/whole-wave-register-spill.ll (+3-3)
``````````diff
diff --git a/llvm/lib/Target/AMDGPU/GCNRegPressure.cpp b/llvm/lib/Target/AMDGPU/GCNRegPressure.cpp
index 53617e89af757..8befd1c614b1e 100644
--- a/llvm/lib/Target/AMDGPU/GCNRegPressure.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNRegPressure.cpp
@@ -373,9 +373,8 @@ static LaneBitmask findUseBetween(unsigned Reg, LaneBitmask LastUseMask,
GCNRPTarget::GCNRPTarget(const MachineFunction &MF, const GCNRegPressure &RP)
: GCNRPTarget(RP, MF) {
- const Function &F = MF.getFunction();
const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
- setTarget(ST.getMaxNumSGPRs(F), ST.getMaxNumVGPRs(F));
+ setTarget(ST.getMaxNumSGPRs(MF), ST.getMaxNumVGPRs(MF));
}
GCNRPTarget::GCNRPTarget(unsigned NumSGPRs, unsigned NumVGPRs,
diff --git a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
index 0ac656a3e02a7..256454b793b34 100644
--- a/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSchedStrategy.cpp
@@ -1389,7 +1389,8 @@ bool RewriteMFMAFormStage::initGCNSchedStage() {
RegionsWithExcessArchVGPR.reset();
for (unsigned Region = 0; Region < DAG.Regions.size(); Region++) {
GCNRegPressure PressureBefore = DAG.Pressure[Region];
- if (PressureBefore.getArchVGPRNum() > ST.getAddressableNumArchVGPRs())
+ if (PressureBefore.getArchVGPRNum() >
+ DAG.RegClassInfo->getNumAllocatableRegs(&AMDGPU::VGPR_32RegClass))
RegionsWithExcessArchVGPR[Region] = true;
}
@@ -2959,11 +2960,9 @@ unsigned PreRARematStage::getStageTargetOccupancy() const {
}
bool PreRARematStage::setObjective() {
- const Function &F = MF.getFunction();
-
// Set up "spilling targets" for all regions.
- unsigned MaxSGPRs = ST.getMaxNumSGPRs(F);
- unsigned MaxVGPRs = ST.getMaxNumVGPRs(F);
+ unsigned MaxSGPRs = ST.getMaxNumSGPRs(MF);
+ unsigned MaxVGPRs = ST.getMaxNumVGPRs(MF);
bool HasVectorRegisterExcess = false;
for (unsigned I = 0, E = DAG.Regions.size(); I != E; ++I) {
const GCNRegPressure &RP = DAG.Pressure[I];
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
index 1c0e718bd8d97..c47565ef9b1b7 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.cpp
@@ -53,6 +53,18 @@ static cl::opt<unsigned>
cl::desc("Number of addresses from which to enable MIMG NSA."),
cl::init(2), cl::Hidden);
+static cl::opt<unsigned> StressVGPRLimit(
+ "amdgpu-stress-vgpr", cl::Hidden, cl::init(0),
+ cl::desc("Limit VGPRs to N arch registers"));
+
+static cl::opt<unsigned> StressAGPRLimit(
+ "amdgpu-stress-agpr", cl::Hidden, cl::init(0),
+ cl::desc("Limit AGPRs to N registers"));
+
+static cl::opt<unsigned> StressSGPRLimit(
+ "amdgpu-stress-sgpr", cl::Hidden, cl::init(0),
+ cl::desc("Limit SGPRs to N registers"));
+
GCNSubtarget::~GCNSubtarget() = default;
static AMDGPUSubtarget::Generation computeDefaultGeneration(const Triple &TT) {
@@ -538,38 +550,9 @@ unsigned GCNSubtarget::getBaseMaxNumSGPRs(
unsigned MaxNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, false);
unsigned MaxAddressableNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, true);
- // Check if maximum number of SGPRs was explicitly requested using
- // "amdgpu-num-sgpr" attribute.
- unsigned Requested =
- F.getFnAttributeAsParsedInteger("amdgpu-num-sgpr", MaxNumSGPRs);
-
- if (Requested != MaxNumSGPRs) {
- // Make sure requested value does not violate subtarget's specifications.
- if (Requested && (Requested <= ReservedNumSGPRs))
- Requested = 0;
-
- // If more SGPRs are required to support the input user/system SGPRs,
- // increase to accommodate them.
- //
- // FIXME: This really ends up using the requested number of SGPRs + number
- // of reserved special registers in total. Theoretically you could re-use
- // the last input registers for these special registers, but this would
- // require a lot of complexity to deal with the weird aliasing.
- unsigned InputNumSGPRs = PreloadedSGPRs;
- if (Requested && Requested < InputNumSGPRs)
- Requested = InputNumSGPRs;
-
- // Make sure requested value is compatible with values implied by
- // default/requested minimum/maximum number of waves per execution unit.
- if (Requested && Requested > getMaxNumSGPRs(WavesPerEU.first, false))
- Requested = 0;
- if (WavesPerEU.second && Requested &&
- Requested < getMinNumSGPRs(WavesPerEU.second))
- Requested = 0;
-
- if (Requested)
- MaxNumSGPRs = Requested;
- }
+ // Stress test: override SGPR limit.
+ if (StressSGPRLimit.getNumOccurrences())
+ MaxNumSGPRs = StressSGPRLimit;
if (hasSGPRInitBug())
MaxNumSGPRs = AMDGPU::IsaInfo::FIXED_NUM_SGPRS_FOR_INIT_BUG;
@@ -609,27 +592,19 @@ unsigned GCNSubtarget::getMaxNumPreloadedSGPRs() const {
return MaxUserSGPRs + MaxSystemSGPRs + SyntheticSGPRs;
}
-unsigned GCNSubtarget::getMaxNumSGPRs(const Function &F) const {
- return getBaseMaxNumSGPRs(F, getWavesPerEU(F), getMaxNumPreloadedSGPRs(),
- getReservedNumSGPRs(F));
+unsigned GCNSubtarget::getMaxNumVGPRs(const Function &F) const {
+ auto [VGPRs, AGPRs] = getMaxNumVectorRegs(F);
+ // On gfx90a+ VGPRs and AGPRs share a unified register file.
+ // On gfx908 they are independent, so only VGPRs count toward the budget.
+ return hasGFX90AInsts() ? VGPRs + AGPRs : VGPRs;
}
-unsigned GCNSubtarget::getBaseMaxNumVGPRs(
- const Function &F, std::pair<unsigned, unsigned> NumVGPRBounds) const {
- const auto [Min, Max] = NumVGPRBounds;
-
- // Check if maximum number of VGPRs was explicitly requested using
- // "amdgpu-num-vgpr" attribute.
-
- unsigned Requested = F.getFnAttributeAsParsedInteger("amdgpu-num-vgpr", Max);
- if (Requested != Max && hasGFX90AInsts())
- Requested *= 2;
-
- // Make sure requested value is inside the range of possible VGPR usage.
- return std::clamp(Requested, Min, Max);
+unsigned GCNSubtarget::getMaxNumVGPRs(const MachineFunction &MF) const {
+ return getMaxNumVGPRs(MF.getFunction());
}
-unsigned GCNSubtarget::getMaxNumVGPRs(const Function &F) const {
+std::pair<unsigned, unsigned>
+GCNSubtarget::getMaxNumVectorRegs(const Function &F) const {
// Temporarily check both the attribute and the subtarget feature, until the
// latter is removed.
unsigned DynamicVGPRBlockSize = AMDGPU::getDynamicVGPRBlockSize(F);
@@ -637,18 +612,8 @@ unsigned GCNSubtarget::getMaxNumVGPRs(const Function &F) const {
DynamicVGPRBlockSize = getDynamicVGPRBlockSize();
std::pair<unsigned, unsigned> Waves = getWavesPerEU(F);
- return getBaseMaxNumVGPRs(
- F, {getMinNumVGPRs(Waves.second, DynamicVGPRBlockSize),
- getMaxNumVGPRs(Waves.first, DynamicVGPRBlockSize)});
-}
-
-unsigned GCNSubtarget::getMaxNumVGPRs(const MachineFunction &MF) const {
- return getMaxNumVGPRs(MF.getFunction());
-}
-
-std::pair<unsigned, unsigned>
-GCNSubtarget::getMaxNumVectorRegs(const Function &F) const {
- const unsigned MaxVectorRegs = getMaxNumVGPRs(F);
+ const unsigned MaxVectorRegs = getMaxNumVGPRs(Waves.first,
+ DynamicVGPRBlockSize);
unsigned MaxNumVGPRs = MaxVectorRegs;
unsigned MaxNumAGPRs = 0;
@@ -700,6 +665,12 @@ GCNSubtarget::getMaxNumVectorRegs(const Function &F) const {
MaxNumAGPRs = MaxNumVGPRs = MaxVectorRegs;
}
+ // Stress test: override VGPR/AGPR limits.
+ if (StressVGPRLimit.getNumOccurrences())
+ MaxNumVGPRs = StressVGPRLimit;
+ if (StressAGPRLimit.getNumOccurrences())
+ MaxNumAGPRs = StressAGPRLimit;
+
return std::pair(MaxNumVGPRs, MaxNumAGPRs);
}
diff --git a/llvm/lib/Target/AMDGPU/GCNSubtarget.h b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
index c42ca8e19ef9c..8545961017f54 100644
--- a/llvm/lib/Target/AMDGPU/GCNSubtarget.h
+++ b/llvm/lib/Target/AMDGPU/GCNSubtarget.h
@@ -811,25 +811,9 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
unsigned ReservedNumSGPRs) const;
/// \returns Maximum number of SGPRs that meets number of waves per execution
- /// unit requirement for function \p MF, or number of SGPRs explicitly
- /// requested using "amdgpu-num-sgpr" attribute attached to function \p MF.
- ///
- /// \returns Value that meets number of waves per execution unit requirement
- /// if explicitly requested value cannot be converted to integer, violates
- /// subtarget's specifications, or does not meet number of waves per execution
- /// unit requirement.
+ /// unit requirement for function \p MF.
unsigned getMaxNumSGPRs(const MachineFunction &MF) const;
- /// \returns Maximum number of SGPRs that meets number of waves per execution
- /// unit requirement for function \p F, or number of SGPRs explicitly
- /// requested using "amdgpu-num-sgpr" attribute attached to function \p F.
- ///
- /// \returns Value that meets number of waves per execution unit requirement
- /// if explicitly requested value cannot be converted to integer, violates
- /// subtarget's specifications, or does not meet number of waves per execution
- /// unit requirement.
- unsigned getMaxNumSGPRs(const Function &F) const;
-
/// \returns VGPR allocation granularity supported by the subtarget.
unsigned getVGPRAllocGranule(unsigned DynamicVGPRBlockSize) const {
return AMDGPU::IsaInfo::getVGPRAllocGranule(*this, DynamicVGPRBlockSize);
@@ -872,36 +856,14 @@ class GCNSubtarget final : public AMDGPUGenSubtargetInfo,
DynamicVGPRBlockSize);
}
- /// \returns max num VGPRs. This is the common utility function
- /// called by MachineFunction and Function variants of getMaxNumVGPRs.
- unsigned
- getBaseMaxNumVGPRs(const Function &F,
- std::pair<unsigned, unsigned> NumVGPRBounds) const;
-
- /// \returns Maximum number of VGPRs that meets number of waves per execution
- /// unit requirement for function \p F, or number of VGPRs explicitly
- /// requested using "amdgpu-num-vgpr" attribute attached to function \p F.
- ///
- /// \returns Value that meets number of waves per execution unit requirement
- /// if explicitly requested value cannot be converted to integer, violates
- /// subtarget's specifications, or does not meet number of waves per execution
- /// unit requirement.
- unsigned getMaxNumVGPRs(const Function &F) const;
-
- unsigned getMaxNumAGPRs(const Function &F) const { return getMaxNumVGPRs(F); }
-
/// Return a pair of maximum numbers of VGPRs and AGPRs that meet the number
- /// of waves per execution unit required for the function \p MF.
+ /// of waves per execution unit required for the function \p F.
std::pair<unsigned, unsigned> getMaxNumVectorRegs(const Function &F) const;
- /// \returns Maximum number of VGPRs that meets number of waves per execution
- /// unit requirement for function \p MF, or number of VGPRs explicitly
- /// requested using "amdgpu-num-vgpr" attribute attached to function \p MF.
- ///
- /// \returns Value that meets number of waves per execution unit requirement
- /// if explicitly requested value cannot be converted to integer, violates
- /// subtarget's specifications, or does not meet number of waves per execution
- /// unit requirement.
+ /// \returns Total vector register budget (VGPRs + AGPRs) for function \p F.
+ unsigned getMaxNumVGPRs(const Function &F) const;
+
+ /// \returns Total vector register budget (VGPRs + AGPRs) for function \p MF.
unsigned getMaxNumVGPRs(const MachineFunction &MF) const;
bool isWave32() const { return getWavefrontSize() == 32; }
diff --git a/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp b/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
index 01220a701a453..3bc24f2c363b3 100644
--- a/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
@@ -40,18 +40,6 @@ static cl::opt<bool> EnableSpillCFISavedRegs(
cl::desc("Enable spilling the registers required for CFI emission"),
cl::ReallyHidden, cl::init(false), cl::ZeroOrMore);
-static cl::opt<unsigned> StressVGPRLimit(
- "amdgpu-stress-vgpr", cl::Hidden, cl::init(0),
- cl::desc("Limit VGPRs to N registers by reserving the rest"));
-
-static cl::opt<unsigned> StressAGPRLimit(
- "amdgpu-stress-agpr", cl::Hidden, cl::init(0),
- cl::desc("Limit AGPRs to N registers by reserving the rest"));
-
-static cl::opt<unsigned> StressSGPRLimit(
- "amdgpu-stress-sgpr", cl::Hidden, cl::init(0),
- cl::desc("Limit SGPRs to N registers by reserving the rest"));
-
std::array<std::vector<int16_t>, 32> SIRegisterInfo::RegSplitParts;
std::array<std::array<uint16_t, 32>, 9> SIRegisterInfo::SubRegFromChannelTable;
@@ -656,8 +644,6 @@ BitVector SIRegisterInfo::getReservedRegs(const MachineFunction &MF) const {
// Reserve SGPRs.
//
unsigned MaxNumSGPRs = ST.getMaxNumSGPRs(MF);
- if (StressSGPRLimit.getNumOccurrences() && StressSGPRLimit < MaxNumSGPRs)
- MaxNumSGPRs = StressSGPRLimit;
unsigned TotalNumSGPRs = AMDGPU::SGPR_32RegClass.getNumRegs();
for (const TargetRegisterClass &RC : regclasses()) {
if (RC.isBaseClass() && isSGPRClass(&RC)) {
@@ -715,12 +701,6 @@ BitVector SIRegisterInfo::getReservedRegs(const MachineFunction &MF) const {
//
auto [MaxNumVGPRs, MaxNumAGPRs] = ST.getMaxNumVectorRegs(MF.getFunction());
- // Stress test: override VGPR/AGPR limits.
- if (StressVGPRLimit.getNumOccurrences() && StressVGPRLimit < MaxNumVGPRs)
- MaxNumVGPRs = StressVGPRLimit;
- if (StressAGPRLimit.getNumOccurrences() && StressAGPRLimit < MaxNumAGPRs)
- MaxNumAGPRs = StressAGPRLimit;
-
for (const TargetRegisterClass &RC : regclasses()) {
if (RC.isBaseClass() && isVGPRClass(&RC)) {
unsigned NumRegs = divideCeil(getRegSizeInBits(RC), 32);
diff --git a/llvm/test/CodeGen/AMDGPU/agpr-remat.ll b/llvm/test/CodeGen/AMDGPU/agpr-remat.ll
index ede7ce390e159..79956b7ac4629 100644
--- a/llvm/test/CodeGen/AMDGPU/agpr-remat.ll
+++ b/llvm/test/CodeGen/AMDGPU/agpr-remat.ll
@@ -1,5 +1,5 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mtriple=amdgpu9.08 < %s | FileCheck -enable-var-scope -check-prefixes=GFX908 %s
+; RUN: llc -mtriple=amdgpu9.08 -amdgpu-stress-vgpr=8 -amdgpu-stress-agpr=8 < %s | FileCheck -enable-var-scope -check-prefixes=GFX908 %s
; Make sure there are no v_accvgpr_read_b32 copying back and forth
; between AGPR and VGPR.
@@ -48,4 +48,4 @@ define void @remat_regcopy_avoids_spill(i32 %v0, i32 %v1, i32 %v2, i32 %v3, i32
ret void
}
-attributes #1 = { nounwind "amdgpu-num-vgpr"="8" }
+attributes #1 = { nounwind }
diff --git a/llvm/test/CodeGen/AMDGPU/attr-amdgpu-num-sgpr.ll b/llvm/test/CodeGen/AMDGPU/attr-amdgpu-num-sgpr.ll
deleted file mode 100644
index d3e56054a7128..0000000000000
--- a/llvm/test/CodeGen/AMDGPU/attr-amdgpu-num-sgpr.ll
+++ /dev/null
@@ -1,131 +0,0 @@
-; RUN: llc -mtriple=amdgpu8.03--amdhsa < %s | FileCheck -check-prefix=ALL %s
-
-; FIXME: Vectorization can increase required SGPR count beyond limit.
-
-; ALL-LABEL: {{^}}max_10_sgprs:
-
-; ALL: SGPRBlocks: 2
-; ALL: NumSGPRsForWavesPerEU: 24
-define amdgpu_kernel void @max_10_sgprs() #0 {
- %one = load volatile i32, ptr addrspace(4) poison
- %two = load volatile i32, ptr addrspace(4) poison
- %three = load volatile i32, ptr addrspace(4) poison
- %four = load volatile i32, ptr addrspace(4) poison
- %five = load volatile i32, ptr addrspace(4) poison
- %six = load volatile i32, ptr addrspace(4) poison
- %seven = load volatile i32, ptr addrspace(4) poison
- %eight = load volatile i32, ptr addrspace(4) poison
- %nine = load volatile i32, ptr addrspace(4) poison
- %ten = load volatile i32, ptr addrspace(4) poison
- %eleven = load volatile i32, ptr addrspace(4) poison
- call void asm sideeffect "", "s,s,s,s,s,s,s,s,s,s"(i32 %one, i32 %two, i32 %three, i32 %four, i32 %five, i32 %six, i32 %seven, i32 %eight, i32 %nine, i32 %ten)
- store volatile i32 %one, ptr addrspace(1) poison
- store volatile i32 %two, ptr addrspace(1) poison
- store volatile i32 %three, ptr addrspace(1) poison
- store volatile i32 %four, ptr addrspace(1) poison
- store volatile i32 %five, ptr addrspace(1) poison
- store volatile i32 %six, ptr addrspace(1) poison
- store volatile i32 %seven, ptr addrspace(1) poison
- store volatile i32 %eight, ptr addrspace(1) poison
- store volatile i32 %nine, ptr addrspace(1) poison
- store volatile i32 %ten, ptr addrspace(1) poison
- store volatile i32 %eleven, ptr addrspace(1) poison
- ret void
-}
-
-; private resource: 4
-; scratch wave offset: 1
-; workgroup ids: 3
-; dispatch id: 2
-; queue ptr: 2
-; flat scratch init: 2
-; ---------------------
-; total: 14
-
-; + reserved vcc = 16
-
-; Because we can't handle re-using the last few input registers as the
-; special vcc etc. registers (as well as decide to not use the unused
-; features when the number of registers is frozen), this ends up using
-; more than expected.
-
-; XALL-LABEL: {{^}}max_12_sgprs_14_input_sgprs:
-; XTOSGPR: SGPRBlocks: 1
-; XTOSGPR: NumSGPRsForWavesPerEU: 16
-
-; This test case is disabled: When calculating the spillslot addresses AMDGPU
-; creates an extra vreg to save/restore m0 which in a point of maximum register
-; pressure would trigger an endless loop; the compiler aborts earlier with
-; "Incomplete scavenging after 2nd pass" in practice.
-;define amdgpu_kernel void @max_12_sgprs_14_input_sgprs(ptr addrspace(1) %out1,
-; ptr addrspace(1) %out2,
-; ptr addrspace(1) %out3,
-; ptr addrspace(1) %out4,
-; i32 %one, i32 %two, i32 %three, i32 %four) #2 {
-; %x.0 = call i32 @llvm.amdgcn.workgroup.id.x()
-; %x.1 = call i32 @llvm.amdgcn.workgroup.id.y()
-; %x.2 = call i32 @llvm.amdgcn.workgroup.id.z()
-; %x.3 = call i64 @llvm.amdgcn.dispatch.id()
-; %x.4 = call ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
-; %x.5 = call ptr addrspace(4) @llvm.amdgcn.queue.ptr()
-; store volatile i32 0, ptr poison
-; br label %stores
-;
-;stores:
-; store volatile i32 %x.0, ptr addrspace(1) poison
-; store volatile i32 %x.0, ptr addrspace(1) poison
-; store volatile i32 %x.0, ptr addrspace(1) poison
-; store volatile i64 %x.3, ptr addrspace(1) poison
-; store volatile ptr addrspace(4) %x.4, ptr addrspace(1) poison
-; store volatile ptr addrspace(4) %x.5, ptr addrspace(1) poison
-;
-; store i32 %one, ptr addrspace(1) %out1
-; store i32 %two, ptr addrspace(1) %out2
-; store i32 %three, ptr addrspace(1) %out3
-; store i32 %four, ptr addrspace(1) %out4
-; ret void
-;}
-
-; The following test is commented out for now; http://llvm.org/PR31230
-; XALL-LABEL: max_12_sgprs_12_input_sgprs{{$}}
-; ; Make sure copies for input buffer are not clobbered. This requires
-; ; swapping the order the registers are copied from what normally
-; ; happens.
-
-; XALL: SGPRBlocks: 2
-; XALL: NumSGPRsForWavesPerEU: 18
-;define amdgpu_kernel void @max_12_sgprs_12_input_sgprs(ptr addrspace(1) %out1,
-; ptr addrspace(1) %out2,
-; ptr addrspace(1) %out3,
-; ptr addrspace(1) %out4,
-; i32 %one, i32 %two, i32 %three, i32 %four) #2 {
-; store volatile i32 0, ptr poison
-; %x.0 = call i32 @llvm.amdgcn.workgroup.id.x()
-; store volatile i32 %x.0, ptr addrspace(1) poison
-; %x.1 = call i32 @llvm.amdgcn.workgroup.id.y()
-; store volatile i32 %x.0, ptr addrspace(1) poison
-; %x.2 = call i32 @llvm.amdgcn.workgroup.id.z()
-; store volatile i32 %x.0, ptr addrspace(1) poison
-; %x.3 = call i64 @llvm.amdgcn.dispatch.id()
-; store volatile i64 %x.3, ptr addrspace(1) poison
-; %x.4 = call ptr addrspace(4) @llvm.amdgcn.dispatch.ptr()
-; store volatile ptr addrspace(4) %x.4, ptr addrspace(1) poison
-;
-; store i32 %one, ptr addrspace(1) %out1
-; store i32 %two, ptr addrspace(1) %out2
-; store i32 %three, ptr addrspace(1) %out3
-; store i32 %four, ptr addrspace(1) %out4
-; ret void
-;}
-
-declare i32 @llvm.amdgcn.workgroup.id.x() #1
-declare i32 @llvm.amdgcn.workgroup.id.y() #1
-declare i32 @llvm.amdgcn.workgroup.id.z() #1
-declare i64 @llvm.amdgcn.dispatch.id() #1
-declare ptr addrspace(4) @llvm.amdgcn.dispatch.ptr() #1
-declare ptr addrspace(4) @llvm.amdgcn.queue.ptr() #1
-
-attributes #0 = { n...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/213924
More information about the llvm-commits
mailing list