[llvm-branch-commits] [llvm] [CodeGen][AMDGPU] Add opt-in partial SGPR spills (PR #225220)
Yaxun Liu via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Mon Sep 21 15:46:05 PDT 2026
https://github.com/yxsamliu updated https://github.com/llvm/llvm-project/pull/225220
>From 1cf089a54dea4ff088aad5ec20e64c0535309d2d Mon Sep 17 00:00:00 2001
From: "Yaxun (Sam) Liu" <yaxun.liu at amd.com>
Date: Mon, 21 Sep 2026 15:12:10 +0000
Subject: [PATCH] [CodeGen][AMDGPU] Add opt-in partial SGPR spills
A wide scalar load can have just one word live after its early uses. A
buffer resource can also be split while its fields are being
constructed. Full-width spills save words that are not needed at those
points.
Add a target hook to save selected lanes at their original spill-slot
offsets while preserving the other words. Use it for AMDGPU SGPR tuples,
protect partially written slots from whole-value spill optimizations,
and remove unchanged word stores after matching reloads.
Keep this behind `-enable-partial-spills`, disabled by default.
Unsupported targets and register classes retain their existing spill
policy. Tests based on loaded tuples and buffer resources let the
allocator generate the split relationships from LLVM IR.
---
llvm/include/llvm/CodeGen/TargetInstrInfo.h | 23 ++
llvm/lib/CodeGen/InlineSpiller.cpp | 150 ++++++--
llvm/lib/Target/AMDGPU/SIInstrInfo.cpp | 62 +++
llvm/lib/Target/AMDGPU/SIInstrInfo.h | 8 +
llvm/lib/Target/AMDGPU/SIInstructions.td | 10 +
llvm/lib/Target/AMDGPU/SILowerSGPRSpills.cpp | 113 +++++-
llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp | 35 +-
.../CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll | 33 ++
.../CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll | 17 +
llvm/test/CodeGen/AMDGPU/dead_bundle.mir | 1 +
.../greedy-alloc-fail-sgpr1024-spill.mir | 29 ++
...nfloop-subrange-spill-inspect-subrange.mir | 30 ++
.../partial-sgpr-spill-buffer-resource.ll | 57 +++
.../AMDGPU/partial-sgpr-spill-cleanup.mir | 353 ++++++++++++++++++
.../AMDGPU/partial-sgpr-spill-loaded-tuple.ll | 56 +++
.../AMDGPU/partial-sgpr-spill-offset.mir | 169 +++++++++
.../AMDGPU/partial-sgpr-spill-scavenge.mir | 90 +++++
.../ra-inserted-scalar-instructions.mir | 21 ++
.../ran-out-of-sgprs-allocation-failure.mir | 19 +
.../CodeGen/AMDGPU/spill-scavenge-offset.ll | 23 ++
.../CodeGen/AMDGPU/splitkit-copy-bundle.mir | 27 ++
.../AMDGPU/splitkit-nolivesubranges.mir | 15 +
llvm/test/CodeGen/AMDGPU/splitkit.mir | 42 +++
.../AMDGPU/tuple-allocation-failure.ll | 2 +
llvm/test/CodeGen/X86/fp16-spill.ll | 1 +
llvm/test/CodeGen/X86/hoist-spill-lpad.ll | 1 +
26 files changed, 1360 insertions(+), 27 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-buffer-resource.ll
create mode 100644 llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-cleanup.mir
create mode 100644 llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-loaded-tuple.ll
create mode 100644 llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-offset.mir
create mode 100644 llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-scavenge.mir
diff --git a/llvm/include/llvm/CodeGen/TargetInstrInfo.h b/llvm/include/llvm/CodeGen/TargetInstrInfo.h
index 9df684096522d..05bfc35382df8 100644
--- a/llvm/include/llvm/CodeGen/TargetInstrInfo.h
+++ b/llvm/include/llvm/CodeGen/TargetInstrInfo.h
@@ -1224,6 +1224,29 @@ class LLVM_ABI TargetInstrInfo : public MCInstrInfo {
"TargetInstrInfo::storeRegToStackSlot!");
}
+ /// Whether RC supports lane-selective ordinary spills. Targets opt in by
+ /// register class so callers can protect shared spill slots before splitting.
+ /// Unsupported classes retain the existing full-register spill policy.
+ /// The emission hook may still reject a particular lane mask without
+ /// mutation.
+ virtual bool supportsPartialSpill(const TargetRegisterClass *RC) const {
+ return false;
+ }
+
+ /// Store only Lanes of SrcReg in the ordinary full-register spill layout.
+ /// The other parts of FrameIndex must be preserved. Return false without
+ /// modifying the function if the selected lanes cannot be stored directly.
+ /// Lanes describes target subregister lanes, not byte offsets. This hook is
+ /// for ordinary spills, not callee-save/unwind operations.
+ virtual bool storeRegToStackSlotPartial(MachineBasicBlock &MBB,
+ MachineBasicBlock::iterator MI,
+ Register SrcReg, bool IsKill,
+ int FrameIndex,
+ const TargetRegisterClass *RC,
+ LaneBitmask Lanes) const {
+ return false;
+ }
+
/// Load the specified register of the given register class from the specified
/// stack frame index. The load instruction is to be added to the given
/// machine basic block before the specified machine instruction. If \p
diff --git a/llvm/lib/CodeGen/InlineSpiller.cpp b/llvm/lib/CodeGen/InlineSpiller.cpp
index f3682a2e24808..c5c1a40f41f46 100644
--- a/llvm/lib/CodeGen/InlineSpiller.cpp
+++ b/llvm/lib/CodeGen/InlineSpiller.cpp
@@ -75,6 +75,15 @@ RestrictStatepointRemat("restrict-statepoint-remat",
cl::init(false), cl::Hidden,
cl::desc("Restrict remat for statepoint operands"));
+static cl::opt<bool> EnablePartialSpills(
+ "enable-partial-spills", cl::init(false), cl::Hidden,
+ cl::desc("Enable lane-selective spills for supported register classes"));
+
+static bool preservePartialSpillSlots(const TargetInstrInfo &TII,
+ const TargetRegisterClass *RC) {
+ return EnablePartialSpills && TII.supportsPartialSpill(RC);
+}
+
namespace {
class HoistSpillHelper : private LiveRangeEdit::Delegate {
MachineFunction &MF;
@@ -102,6 +111,10 @@ class HoistSpillHelper : private LiveRangeEdit::Delegate {
using MergeableSpillsMap =
MapVector<std::pair<int, VNInfo *>, SmallPtrSet<MachineInstr *, 16>>;
MergeableSpillsMap MergeableSpills;
+ // Slots that may acquire partial writers during allocation.
+ SmallSetVector<int, 8> PartialSpillSlots;
+ // Slots with an observed partial writer, which prevents late spill hoisting.
+ SmallSetVector<int, 8> PartialWriteSlots;
/// This is the map from original register to a set containing all its
/// siblings. To hoist a spill to another BB, we need to find out a live
@@ -140,6 +153,14 @@ class HoistSpillHelper : private LiveRangeEdit::Delegate {
Register Original);
bool rmFromMergeableSpills(MachineInstr &Spill, int StackSlot);
void hoistAllSpills();
+ void markPartialSpillSlot(int Slot) { PartialSpillSlots.insert(Slot); }
+ void markPartialWrite(int Slot) {
+ markPartialSpillSlot(Slot);
+ PartialWriteSlots.insert(Slot);
+ }
+ bool isPartialSpillSlot(int Slot) const {
+ return PartialSpillSlots.contains(Slot);
+ }
void LRE_WillShrinkVirtReg(Register) override;
bool LRE_CanEraseVirtReg(Register) override;
void LRE_DidCloneVirtReg(Register, Register) override;
@@ -225,8 +246,11 @@ class InlineSpiller : public Spiller {
bool coalesceStackAccess(MachineInstr *MI, Register Reg);
bool foldMemoryOperand(ArrayRef<std::pair<MachineInstr *, unsigned>>,
MachineInstr *LoadMI = nullptr);
- void insertReload(Register VReg, SlotIndex, MachineBasicBlock::iterator MI);
+ SlotIndex insertReload(Register VReg, SlotIndex,
+ MachineBasicBlock::iterator MI);
void insertSpill(Register VReg, bool isKill, MachineBasicBlock::iterator MI);
+ bool insertPartialSpill(Register VReg, LaneBitmask Lanes,
+ MachineBasicBlock::iterator MI);
void spillAroundUses(Register Reg);
void spillAll();
@@ -440,6 +464,8 @@ bool InlineSpiller::isSibling(Register Reg) {
///
bool InlineSpiller::hoistSpillInsideBB(LiveInterval &SpillLI,
MachineInstr &CopyMI) {
+ if (HSpiller.isPartialSpillSlot(StackSlot))
+ return false;
SlotIndex Idx = LIS.getInstructionIndex(CopyMI);
#ifndef NDEBUG
VNInfo *VNI = SpillLI.getVNInfoAt(Idx.getRegSlot());
@@ -526,6 +552,8 @@ bool InlineSpiller::hoistSpillInsideBB(LiveInterval &SpillLI,
/// eliminateRedundantSpills - SLI:VNI is known to be on the stack. Remove any
/// redundant spills of this value in SLI.reg and sibling copies.
void InlineSpiller::eliminateRedundantSpills(LiveInterval &SLI, VNInfo *VNI) {
+ if (HSpiller.isPartialSpillSlot(StackSlot))
+ return;
assert(VNI && "Missing value");
SmallVector<std::pair<LiveInterval*, VNInfo*>, 8> WorkList;
WorkList.push_back(std::make_pair(&SLI, VNI));
@@ -1244,9 +1272,8 @@ foldMemoryOperand(ArrayRef<std::pair<MachineInstr *, unsigned>> Ops,
return true;
}
-void InlineSpiller::insertReload(Register NewVReg,
- SlotIndex Idx,
- MachineBasicBlock::iterator MI) {
+SlotIndex InlineSpiller::insertReload(Register NewVReg, SlotIndex Idx,
+ MachineBasicBlock::iterator MI) {
MachineBasicBlock &MBB = *MI->getParent();
MachineInstrSpan MIS(MI, &MBB);
@@ -1258,6 +1285,7 @@ void InlineSpiller::insertReload(Register NewVReg,
LLVM_DEBUG(dumpMachineInstrRangeWithSlotIndex(MIS.begin(), MI, LIS, "reload",
NewVReg));
++NumReloads;
+ return LIS.getInstructionIndex(*MIS.begin()).getBaseIndex();
}
/// Check if \p Def fully defines a VReg with an undefined value.
@@ -1311,10 +1339,27 @@ void InlineSpiller::insertSpill(Register NewVReg, bool isKill,
HSpiller.addToMergeableSpills(*Spill, StackSlot, Original);
}
+bool InlineSpiller::insertPartialSpill(Register VReg, LaneBitmask Lanes,
+ MachineBasicBlock::iterator MI) {
+ MachineBasicBlock &MBB = *MI->getParent();
+ MachineInstrSpan MIS(MI, &MBB);
+ MachineBasicBlock::iterator SpillBefore = std::next(MI);
+ if (!TII.storeRegToStackSlotPartial(MBB, SpillBefore, VReg, true, StackSlot,
+ MRI.getRegClass(VReg), Lanes))
+ return false;
+ MachineBasicBlock::iterator First = std::next(MI);
+ LIS.InsertMachineInstrRangeInMaps(First, MIS.end());
+ for (const MachineInstr &Spill : make_range(First, MIS.end()))
+ getVDefInterval(Spill, LIS);
+ ++NumSpills;
+ return true;
+}
+
/// spillAroundUses - insert spill code around each use of Reg.
void InlineSpiller::spillAroundUses(Register Reg) {
LLVM_DEBUG(dbgs() << "spillAroundUses " << printReg(Reg) << '\n');
LiveInterval &OldLI = LIS.getInterval(Reg);
+ LaneBitmask MaxLanes = MRI.getMaxLaneMaskForVReg(Reg);
// Iterate over instructions using Reg.
for (MachineInstr &MI : llvm::make_early_inc_range(MRI.reg_bundles(Reg))) {
@@ -1342,6 +1387,23 @@ void InlineSpiller::spillAroundUses(Register Reg) {
// Analyze instruction.
SmallVector<std::pair<MachineInstr*, unsigned>, 8> Ops;
VirtRegInfo RI = AnalyzeVirtRegInBundle(MI, Reg, &Ops);
+ bool HasLiveDef = any_of(Ops, [](const auto &OpPair) {
+ const MachineOperand &MO = OpPair.first->getOperand(OpPair.second);
+ return MO.isDef() && !MO.isDead();
+ });
+ // A partial definition must preserve slot words used by wider siblings.
+ // Store only the defined lanes if possible; otherwise reload the slot
+ // before the definition so a full spill preserves the remaining lanes.
+ bool NeedsPartialSpill =
+ preservePartialSpillSlots(TII, MRI.getRegClass(Reg)) && HasLiveDef &&
+ RI.Writes && !RI.Reads && isRealSpill(MI);
+ if (NeedsPartialSpill) {
+ LaneBitmask Defined =
+ AnalyzeVirtRegLanesInBundle(MI, Reg, MRI, TRI).second;
+ NeedsPartialSpill = (MaxLanes & ~Defined).any();
+ }
+ if (NeedsPartialSpill)
+ HSpiller.markPartialWrite(StackSlot);
// Find the slot index where this instruction reads and writes OldLI.
// This is usually the def slot, except for tied early clobbers.
@@ -1350,7 +1412,8 @@ void InlineSpiller::spillAroundUses(Register Reg) {
if (SlotIndex::isSameInstr(Idx, VNI->def))
Idx = VNI->def;
- // Check for a sibling copy.
+ // Copies within this spill group share storage. The store-moving and
+ // store-elimination helpers enforce the partial-slot guard separately.
Register SibReg = isCopyOfBundle(MI, Reg, TII);
if (SibReg && isSibling(SibReg)) {
// This may actually be a copy between snippets.
@@ -1375,35 +1438,53 @@ void InlineSpiller::spillAroundUses(Register Reg) {
}
// Attempt to fold memory ops.
- if (foldMemoryOperand(Ops))
+ if (!NeedsPartialSpill && foldMemoryOperand(Ops))
continue;
+ // Do not introduce uses of dead definitions in a partial-copy bundle.
+ LaneBitmask LiveDefined = LaneBitmask::getNone();
+ if (NeedsPartialSpill) {
+ for (const auto &OpPair : Ops) {
+ const MachineOperand &MO = OpPair.first->getOperand(OpPair.second);
+ if (MO.isDef() && !MO.isDead())
+ LiveDefined |= MO.getSubReg()
+ ? TRI.getSubRegIndexLaneMask(MO.getSubReg())
+ : MaxLanes;
+ }
+ }
+
// Create a new virtual register for spill/fill.
// FIXME: Infer regclass from instruction alone.
Register NewVReg = Edit->createFrom(Reg);
-
- if (RI.Reads)
- insertReload(NewVReg, Idx, &MI);
+ bool StoredPartial = NeedsPartialSpill && LiveDefined.any() &&
+ insertPartialSpill(NewVReg, LiveDefined, &MI);
+ bool NeedsPreservingReload = NeedsPartialSpill && !StoredPartial;
+
+ if (RI.Reads || NeedsPreservingReload) {
+ SlotIndex ReloadIdx = insertReload(NewVReg, Idx, &MI);
+ // The preservation read precedes the old partial definition, so the
+ // original register interval does not cover this stack-slot use.
+ if (NeedsPreservingReload)
+ StackInt->addSegment(
+ LiveRange::Segment(ReloadIdx, Idx, StackInt->getValNumInfo(0)));
+ }
// Rewrite instruction operands.
- bool hasLiveDef = false;
for (const auto &OpPair : Ops) {
MachineOperand &MO = OpPair.first->getOperand(OpPair.second);
MO.setReg(NewVReg);
+ if (NeedsPreservingReload && MO.isDef() && MO.getSubReg())
+ MO.setIsUndef(false);
if (MO.isUse()) {
if (!OpPair.first->isRegTiedToDefOperand(OpPair.second))
MO.setIsKill();
- } else {
- if (!MO.isDead())
- hasLiveDef = true;
}
}
LLVM_DEBUG(dbgs() << "\trewrite: " << Idx << '\t' << MI << '\n');
// FIXME: Use a second vreg if instruction has no tied ops.
- if (RI.Writes)
- if (hasLiveDef)
- insertSpill(NewVReg, true, &MI);
+ if (RI.Writes && HasLiveDef && !StoredPartial)
+ insertSpill(NewVReg, true, &MI);
}
}
@@ -1420,6 +1501,11 @@ void InlineSpiller::spillAll() {
if (Original != Edit->getReg())
VRM.assignVirt2StackSlot(Edit->getReg(), StackSlot);
+ // Reserve conservative spill handling before any sibling can be optimized.
+ // Later splits of an opted-in class may introduce partial writers.
+ if (preservePartialSpillSlots(TII, MRI.getRegClass(Original)))
+ HSpiller.markPartialSpillSlot(StackSlot);
+
assert(StackInt->getNumValNums() == 1 && "Bad stack interval values");
for (Register Reg : RegsToSpill)
StackInt->MergeSegmentsInAsValue(LIS.getInterval(Reg),
@@ -1539,19 +1625,37 @@ bool HoistSpillHelper::isSpillCandBB(LiveInterval &OrigLI, VNInfo &OrigVNI,
SmallSetVector<Register, 16> &Siblings = Virt2SiblingsMap[OrigReg];
assert(OrigLI.getVNInfoAt(Idx) == &OrigVNI && "Unexpected VNI");
+ const TargetRegisterClass *OrigRC = MRI.getRegClass(OrigReg);
+ bool PreserveSlot = preservePartialSpillSlots(TII, OrigRC);
for (const Register &SibReg : Siblings) {
LiveInterval &LI = LIS.getInterval(SibReg);
if (!LI.getVNInfoAt(Idx))
continue;
+ if (PreserveSlot &&
+ TRI.getSpillSize(*MRI.getRegClass(SibReg)) != TRI.getSpillSize(*OrigRC))
+ continue;
+
// All of the sub-ranges should be alive at the prospective slot index.
// Otherwise, we might risk storing unrelated / compromised values from some
// sub-registers to the spill slot.
- if (all_of(LI.subranges(), [&](const LiveInterval::SubRange &SR) {
- return SR.getVNInfoAt(Idx) != nullptr;
- })) {
- LiveReg = SibReg;
- return true;
+ LaneBitmask Covered = LaneBitmask::getNone();
+ bool AllSubRangesLive = true;
+ for (const LiveInterval::SubRange &SR : LI.subranges()) {
+ if (!SR.getVNInfoAt(Idx)) {
+ AllSubRangesLive = false;
+ break;
+ }
+ Covered |= SR.LaneMask;
}
+ if (!AllSubRangesLive)
+ continue;
+ // A sparse sibling cannot supply the complete original spill layout.
+ if (PreserveSlot && LI.hasSubRanges() &&
+ Covered != MRI.getMaxLaneMaskForVReg(OrigReg))
+ continue;
+
+ LiveReg = SibReg;
+ return true;
}
return false;
}
@@ -1823,6 +1927,10 @@ void HoistSpillHelper::hoistAllSpills() {
// Each entry in MergeableSpills contains a spill set with equal values.
for (auto &Ent : MergeableSpills) {
int Slot = Ent.first.first;
+ // All spill insertion is complete. Slots that never acquired a partial
+ // writer no longer need the conservative guard used during allocation.
+ if (PartialWriteSlots.contains(Slot))
+ continue;
LiveInterval &OrigLI = *StackSlotToOrigLI[Slot];
VNInfo *OrigVNI = Ent.first.second;
SmallPtrSet<MachineInstr *, 16> &EqValSpills = Ent.second;
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
index ea16b6e886a44..d331f46cd79b1 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.cpp
@@ -1828,6 +1828,63 @@ void SIInstrInfo::storeRegToStackSlot(
false);
}
+bool SIInstrInfo::supportsPartialSpill(const TargetRegisterClass *RC) const {
+ return RI.isSGPRClass(RC) && RI.getSpillSize(*RC) > 4;
+}
+
+bool SIInstrInfo::storeRegToStackSlotPartial(MachineBasicBlock &MBB,
+ MachineBasicBlock::iterator MI,
+ Register SrcReg, bool IsKill,
+ int FrameIndex,
+ const TargetRegisterClass *RC,
+ LaneBitmask Lanes) const {
+ if (!supportsPartialSpill(RC) || !SrcReg.isVirtual())
+ return false;
+
+ MachineFunction &MF = *MBB.getParent();
+ MachineFrameInfo &FrameInfo = MF.getFrameInfo();
+ ArrayRef<int16_t> Parts = RI.getRegSplitParts(RC, 4);
+ if (Parts.empty() ||
+ FrameInfo.getObjectSize(FrameIndex) < RI.getSpillSize(*RC))
+ return false;
+
+ // Validate the complete 32-bit word selection before mutating the function:
+ // a rejected lane mask must leave callers free to use the fallback.
+ SmallVector<unsigned, 8> Selected;
+ LaneBitmask Covered = LaneBitmask::getNone();
+ for (unsigned Part : Parts) {
+ LaneBitmask PartLanes = RI.getSubRegIndexLaneMask(Part);
+ if ((Lanes & PartLanes).none())
+ continue;
+ if ((Lanes & PartLanes) != PartLanes)
+ return false;
+ Selected.push_back(Part);
+ Covered |= PartLanes;
+ }
+ if (Selected.empty() || Covered != Lanes)
+ return false;
+
+ SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
+ if (FrameInfo.getStackID(FrameIndex) == TargetStackID::SGPRSpill)
+ MFI->setHasSpilledSGPRs();
+ const DebugLoc &DL = MBB.findDebugLoc(MI);
+ for (unsigned Part : Selected) {
+ unsigned Offset = RI.getSubRegIdxOffset(Part) / 8;
+ MachineMemOperand *MMO = MF.getMachineMemOperand(
+ MachinePointerInfo::getFixedStack(MF, FrameIndex, Offset),
+ MachineMemOperand::MOStore, 4,
+ commonAlignment(FrameInfo.getObjectAlign(FrameIndex), Offset));
+ BuildMI(MBB, MI, DL, get(AMDGPU::SI_SPILL_S32_SAVE_partial))
+ .addReg(SrcReg, getKillRegState(IsKill && Part == Selected.back()),
+ Part)
+ .addFrameIndex(FrameIndex)
+ .addImm(Offset)
+ .addMemOperand(MMO)
+ .addReg(MFI->getStackPtrOffsetReg(), RegState::Implicit);
+ }
+ return true;
+}
+
void SIInstrInfo::storeRegToStackSlotCFI(MachineBasicBlock &MBB,
MachineBasicBlock::iterator MI,
Register SrcReg, bool isKill,
@@ -10232,6 +10289,11 @@ Register SIInstrInfo::isStoreToStackSlot(const MachineInstr &MI,
if (!MI.mayStore())
return Register();
+ // This query cannot describe a source subregister or a nonzero slot offset.
+ // Do not let whole-value coalescing/elimination consume a partial store.
+ if (MI.getOpcode() == AMDGPU::SI_SPILL_S32_SAVE_partial)
+ return Register();
+
if (isMUBUF(MI) || isVGPRSpill(MI))
return isStackAccess(MI, FrameIndex, MemBytes);
diff --git a/llvm/lib/Target/AMDGPU/SIInstrInfo.h b/llvm/lib/Target/AMDGPU/SIInstrInfo.h
index 48eba6b2a567d..cc9ec9e849eb5 100644
--- a/llvm/lib/Target/AMDGPU/SIInstrInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIInstrInfo.h
@@ -356,6 +356,14 @@ class SIInstrInfo final : public AMDGPUGenInstrInfo {
bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg,
MachineInstr::MIFlag Flags = MachineInstr::NoFlags) const override;
+ bool supportsPartialSpill(const TargetRegisterClass *RC) const override;
+
+ bool storeRegToStackSlotPartial(MachineBasicBlock &MBB,
+ MachineBasicBlock::iterator MI,
+ Register SrcReg, bool IsKill, int FrameIndex,
+ const TargetRegisterClass *RC,
+ LaneBitmask Lanes) const override;
+
void loadRegFromStackSlot(
MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg,
int FrameIndex, const TargetRegisterClass *RC, Register VReg,
diff --git a/llvm/lib/Target/AMDGPU/SIInstructions.td b/llvm/lib/Target/AMDGPU/SIInstructions.td
index eaece88e98525..676770bd4fe91 100644
--- a/llvm/lib/Target/AMDGPU/SIInstructions.td
+++ b/llvm/lib/Target/AMDGPU/SIInstructions.td
@@ -1242,6 +1242,16 @@ def SI_RESTORE_S32_FROM_VGPR : PseudoInstSI <(outs SReg_32:$sdst),
}
} // End Spill = 1, VALU = 1, isConvergent = 1
+// Preserve the original word position when saving one part of a shared slot.
+let UseNamedOperandTable = 1, Spill = 1, SALU = 1, Uses = [EXEC] in {
+ def SI_SPILL_S32_SAVE_partial
+ : PseudoInstSI<(outs), (ins SReg_32:$data, i32imm:$addr,
+ i32imm:$offset)> {
+ let mayStore = 1;
+ let mayLoad = 0;
+ }
+}
+
// VGPR or AGPR spill instructions. In case of AGPR spilling a temp register
// needs to be used and an extra instruction to move between VGPR and AGPR.
// UsesTmp adds to the total size of an expanded spill in this case.
diff --git a/llvm/lib/Target/AMDGPU/SILowerSGPRSpills.cpp b/llvm/lib/Target/AMDGPU/SILowerSGPRSpills.cpp
index d0a57e163d29a..e998eaf408863 100644
--- a/llvm/lib/Target/AMDGPU/SILowerSGPRSpills.cpp
+++ b/llvm/lib/Target/AMDGPU/SILowerSGPRSpills.cpp
@@ -21,17 +21,23 @@
#include "SIMachineFunctionInfo.h"
#include "SIPreAllocateWWMRegs.h"
#include "SISpillUtils.h"
+#include "llvm/ADT/Statistic.h"
#include "llvm/CodeGen/LiveIntervals.h"
#include "llvm/CodeGen/MachineCycleAnalysis.h"
#include "llvm/CodeGen/MachineDominators.h"
#include "llvm/CodeGen/MachineFrameInfo.h"
+#include "llvm/CodeGen/MachineMemOperand.h"
#include "llvm/CodeGen/RegisterScavenging.h"
#include "llvm/InitializePasses.h"
+#include <optional>
using namespace llvm;
#define DEBUG_TYPE "si-lower-sgpr-spills"
+STATISTIC(NumPartialSpillsRemoved,
+ "Number of unchanged partial SGPR spills removed");
+
using MBBVector = SmallVector<MachineBasicBlock *, 4>;
namespace {
@@ -68,6 +74,8 @@ class SILowerSGPRSpills {
MBBVector RestoreBlocks;
MachineBasicBlock *getCycleDomBB(CycleRef C);
+ bool isUnchangedPartialSpill(const MachineInstr &Store) const;
+ bool removeRedundantPartialSpills(MachineFunction &MF);
public:
SILowerSGPRSpills(LiveIntervals *LIS, SlotIndexes *Indexes,
@@ -445,6 +453,109 @@ bool SILowerSGPRSpillsLegacy::runOnMachineFunction(MachineFunction &MF) {
return SILowerSGPRSpills(LIS, Indexes, MDT, MCI).run(MF);
}
+// A partial store is redundant if its value came from the same slot word.
+// Trace physical copies back to a matching reload within this block,
+// stopping at clobbers or possible memory interference. Bound the search
+// to limit compile time; stack liveness alone does not prove redundancy.
+bool SILowerSGPRSpills::isUnchangedPartialSpill(
+ const MachineInstr &Store) const {
+ auto HasVolatileOrAtomicAccess = [](const MachineInstr &MI) {
+ return llvm::any_of(MI.memoperands(), [](const MachineMemOperand *MMO) {
+ return MMO->isVolatile() || MMO->isAtomic();
+ });
+ };
+ const MachineOperand &Data = Store.getOperand(0);
+ if (!Data.getReg().isPhysical() || Data.getSubReg() || Data.isUndef() ||
+ !AMDGPU::SGPR_32RegClass.contains(Data.getReg()) ||
+ HasVolatileOrAtomicAccess(Store))
+ return false;
+ MCRegister Reg = Data.getReg();
+ int FI = TII->getNamedOperand(Store, AMDGPU::OpName::addr)->getIndex();
+ int64_t Offset =
+ TII->getNamedOperand(Store, AMDGPU::OpName::offset)->getImm();
+ unsigned Visited = 0;
+ auto I = Store.getIterator();
+ while (I != Store.getParent()->instr_begin()) {
+ const MachineInstr &MI = *--I;
+ if (MI.isDebugInstr())
+ continue;
+ if (++Visited > 64 || MI.isBundled() || MI.isCall() ||
+ HasVolatileOrAtomicAccess(MI) ||
+ llvm::any_of(MI.operands(),
+ [](const MachineOperand &MO) { return MO.isRegMask(); }))
+ return false;
+
+ if (TII->isSGPRSpill(MI)) {
+ int OtherFI = TII->getNamedOperand(MI, AMDGPU::OpName::addr)->getIndex();
+ if (MI.mayStore() && OtherFI == FI) {
+ if (MI.getOpcode() != AMDGPU::SI_SPILL_S32_SAVE_partial ||
+ TII->getNamedOperand(MI, AMDGPU::OpName::offset)->getImm() ==
+ Offset)
+ return false;
+ }
+ if (MI.mayLoad() && MI.modifiesRegister(Reg, TRI)) {
+ TypeSize Bytes = TypeSize::getZero();
+ Register Loaded = TII->isLoadFromStackSlot(MI, OtherFI, Bytes);
+ if (OtherFI != FI || !Loaded.isPhysical() ||
+ MI.getOperand(0).getSubReg())
+ return false;
+ unsigned Sub = Loaded == Reg ? 0 : TRI->getSubRegIndex(Loaded, Reg);
+ if (Loaded != Reg && !Sub)
+ return false;
+ unsigned LoadedOffset = Sub ? TRI->getSubRegIdxOffset(Sub) / 8 : 0;
+ return Offset == LoadedOffset &&
+ Bytes == TypeSize::getFixed(TRI->getSpillSize(
+ *TRI->getPhysRegBaseClass(Loaded)));
+ }
+ } else if (MI.mayStore() || MI.hasUnmodeledSideEffects()) {
+ return false;
+ }
+
+ if (!MI.modifiesRegister(Reg, TRI))
+ continue;
+ std::optional<DestSourcePair> Copy = TII->isCopyInstr(MI);
+ if (!Copy || !Copy->Source->isReg() || Copy->Source->isUndef() ||
+ Copy->Source->getSubReg() || Copy->Destination->getSubReg() ||
+ !Copy->Source->getReg().isPhysical() ||
+ !Copy->Destination->getReg().isPhysical())
+ return false;
+ MCRegister Dst = Copy->Destination->getReg();
+ unsigned Sub = Dst == Reg ? 0 : TRI->getSubRegIndex(Dst, Reg);
+ if (Dst != Reg && !Sub)
+ return false;
+ Reg = Sub ? TRI->getSubReg(Copy->Source->getReg(), Sub)
+ : Copy->Source->getReg().asMCReg();
+ if (!AMDGPU::SGPR_32RegClass.contains(Reg))
+ return false;
+ }
+ return false;
+}
+
+bool SILowerSGPRSpills::removeRedundantPartialSpills(MachineFunction &MF) {
+ bool Changed = false;
+ for (MachineBasicBlock &MBB : MF) {
+ for (MachineInstr &MI : llvm::make_early_inc_range(MBB)) {
+ if (MI.getOpcode() != AMDGPU::SI_SPILL_S32_SAVE_partial ||
+ MI.isBundled() || !isUnchangedPartialSpill(MI))
+ continue;
+ LLVM_DEBUG(dbgs() << "Removing unchanged partial spill: " << MI);
+ // Removing a physical use can shorten cached register-unit intervals.
+ if (LIS) {
+ for (const MachineOperand &MO : MI.operands()) {
+ if (MO.isReg() && MO.getReg().isPhysical())
+ LIS->removeAllRegUnitsForPhysReg(MO.getReg());
+ }
+ }
+ if (Indexes)
+ Indexes->removeMachineInstrFromMaps(MI);
+ MI.eraseFromParent();
+ ++NumPartialSpillsRemoved;
+ Changed = true;
+ }
+ }
+ return Changed;
+}
+
bool SILowerSGPRSpills::run(MachineFunction &MF) {
const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
TII = ST.getInstrInfo();
@@ -468,7 +579,7 @@ bool SILowerSGPRSpills::run(MachineFunction &MF) {
return false;
}
- bool MadeChange = false;
+ bool MadeChange = removeRedundantPartialSpills(MF);
bool SpilledToVirtVGPRLanes = false;
// TODO: CSR VGPRs will never be spilled to AGPRs. These can probably be
diff --git a/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp b/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
index ccc5928109fa6..b30be851c28a5 100644
--- a/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
+++ b/llvm/lib/Target/AMDGPU/SIRegisterInfo.cpp
@@ -101,6 +101,8 @@ struct SGPRSpillBuilder {
MachineBasicBlock::iterator MI;
ArrayRef<int16_t> SplitParts;
unsigned NumSubRegs;
+ unsigned SpillLane = 0;
+ bool IsPartialStore = false;
bool IsKill;
const DebugLoc &DL;
@@ -145,6 +147,17 @@ struct SGPRSpillBuilder {
const TargetRegisterClass *RC = TRI.getPhysRegBaseClass(SuperReg);
SplitParts = TRI.getRegSplitParts(RC, EltSize);
NumSubRegs = SplitParts.empty() ? 1 : SplitParts.size();
+ // A partial store still uses the original slot's word-to-lane mapping,
+ // even though its source contains only one SGPR.
+ IsPartialStore = MI->getOpcode() == AMDGPU::SI_SPILL_S32_SAVE_partial;
+ if (IsPartialStore) {
+ int64_t Offset = MI->getOperand(2).getImm();
+ assert(NumSubRegs == 1 && Offset >= 0 && Offset % 4 == 0 &&
+ "invalid partial SGPR spill");
+ SpillLane = Offset / 4;
+ assert(SpillLane < (IsWave32 ? 32u : 64u) &&
+ "partial SGPR spill exceeds one wave");
+ }
if (IsWave32) {
ExecReg = AMDGPU::EXEC_LO;
@@ -165,7 +178,9 @@ struct SGPRSpillBuilder {
PerVGPRData Data;
Data.PerVGPR = IsWave32 ? 32 : 64;
Data.NumVGPRs = (NumSubRegs + (Data.PerVGPR - 1)) / Data.PerVGPR;
- Data.VGPRLanes = (1LL << std::min(Data.PerVGPR, NumSubRegs)) - 1LL;
+ Data.VGPRLanes = IsPartialStore
+ ? int64_t(1ULL << SpillLane)
+ : (1LL << std::min(Data.PerVGPR, NumSubRegs)) - 1LL;
return Data;
}
@@ -1311,6 +1326,7 @@ static unsigned getNumSubRegsForSpillOp(const MachineInstr &MI,
case AMDGPU::SI_SPILL_AV64_CFI_SAVE:
case AMDGPU::SI_SPILL_AV64_RESTORE:
return 2;
+ case AMDGPU::SI_SPILL_S32_SAVE_partial:
case AMDGPU::SI_SPILL_S32_SAVE:
case AMDGPU::SI_SPILL_S32_CFI_SAVE:
case AMDGPU::SI_SPILL_S32_RESTORE:
@@ -2185,7 +2201,7 @@ bool SIRegisterInfo::spillSGPR(MachineBasicBlock::iterator MI, int Index,
// VGPR lanes (mapped from spill stack slot) may be shared for SGPR
// spills of different sizes. This accounts for number of VGPR lanes alloted
// equal to the largest SGPR being spilled in them.
- assert(SB.NumSubRegs <= VGPRSpills.size() &&
+ assert(SB.SpillLane + SB.NumSubRegs <= VGPRSpills.size() &&
"Num of SGPRs spilled should be less than or equal to num of "
"the VGPR lanes.");
@@ -2194,7 +2210,7 @@ bool SIRegisterInfo::spillSGPR(MachineBasicBlock::iterator MI, int Index,
SB.NumSubRegs == 1
? SB.SuperReg
: Register(getSubReg(SB.SuperReg, SB.SplitParts[i]));
- SpilledReg Spill = VGPRSpills[i];
+ SpilledReg Spill = VGPRSpills[SB.SpillLane + i];
bool IsFirstSubreg = i == 0;
bool IsLastSubreg = i == SB.NumSubRegs - 1;
@@ -2253,8 +2269,15 @@ bool SIRegisterInfo::spillSGPR(MachineBasicBlock::iterator MI, int Index,
// Per VGPR helper data
auto PVD = SB.getPerVGPRData();
+ // Without a saved EXEC register, the memory helper writes every lane.
+ // Preserve the other slot words in the scavenged VGPR before that write.
+ // This happens after SGPR allocation and needs no wide SGPR temporary.
+ if (SB.IsPartialStore && !SB.SavedExecReg)
+ SB.readWriteTmpVGPR(0, /*IsLoad=*/true);
+
for (unsigned Offset = 0; Offset < PVD.NumVGPRs; ++Offset) {
- RegState TmpVGPRFlags = RegState::Undef;
+ RegState TmpVGPRFlags =
+ SB.IsPartialStore && !SB.SavedExecReg ? RegState{} : RegState::Undef;
// Write sub registers into the VGPR
for (unsigned i = Offset * PVD.PerVGPR,
@@ -2269,7 +2292,7 @@ bool SIRegisterInfo::spillSGPR(MachineBasicBlock::iterator MI, int Index,
BuildMI(*SB.MBB, MI, SB.DL,
SB.TII.get(AMDGPU::SI_SPILL_S32_TO_VGPR), SB.TmpVGPR)
.addReg(SubReg, SubKillState)
- .addImm(i % PVD.PerVGPR)
+ .addImm((SB.SpillLane + i) % PVD.PerVGPR)
.addReg(SB.TmpVGPR, TmpVGPRFlags);
TmpVGPRFlags = {};
@@ -2503,6 +2526,7 @@ bool SIRegisterInfo::eliminateSGPRToVGPRSpillFrameIndex(
case AMDGPU::SI_SPILL_S128_SAVE:
case AMDGPU::SI_SPILL_S96_SAVE:
case AMDGPU::SI_SPILL_S64_SAVE:
+ case AMDGPU::SI_SPILL_S32_SAVE_partial:
case AMDGPU::SI_SPILL_S32_SAVE:
return spillSGPR(MI, FI, RS, Indexes, LIS, true, SpillToPhysVGPRLane,
NeedsCFI);
@@ -2593,6 +2617,7 @@ bool SIRegisterInfo::eliminateFrameIndex(MachineBasicBlock::iterator MI,
case AMDGPU::SI_SPILL_S128_SAVE:
case AMDGPU::SI_SPILL_S96_SAVE:
case AMDGPU::SI_SPILL_S64_SAVE:
+ case AMDGPU::SI_SPILL_S32_SAVE_partial:
case AMDGPU::SI_SPILL_S32_SAVE: {
return spillSGPR(MI, Index, RS, nullptr, nullptr,
FrameInfo.getStackID(Index) == TargetStackID::SGPRSpill,
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
index 47468f3c7c4cf..1b9e101c8478b 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
@@ -1,3 +1,36 @@
+; RUN: llc -mtriple=amdgpu6.00 -enable-partial-spills -verify-machineinstrs=0 -verify-regalloc < %s | FileCheck %s --check-prefix=PARTIAL
+
+; Partial scalar-pair spills omit dead high words. Bound the total lane writes
+; in each changed function while retaining the default instruction checks.
+; PARTIAL-LABEL: bitcast_v32i32_to_v128i8_scalar:
+; PARTIAL-COUNT-198: v_writelane_b32
+; PARTIAL-NOT: v_writelane_b32
+; PARTIAL: .Lfunc_end
+; PARTIAL-LABEL: bitcast_v32f32_to_v128i8_scalar:
+; PARTIAL-COUNT-148: v_writelane_b32
+; PARTIAL-NOT: v_writelane_b32
+; PARTIAL: .Lfunc_end
+; PARTIAL-LABEL: bitcast_v16i64_to_v128i8_scalar:
+; PARTIAL-COUNT-238: v_writelane_b32
+; PARTIAL-NOT: v_writelane_b32
+; PARTIAL: .Lfunc_end
+; PARTIAL-LABEL: bitcast_v16f64_to_v128i8_scalar:
+; PARTIAL-COUNT-135: v_writelane_b32
+; PARTIAL-NOT: v_writelane_b32
+; PARTIAL: .Lfunc_end
+; PARTIAL-LABEL: bitcast_v64bf16_to_v128i8_scalar:
+; PARTIAL-COUNT-282: v_writelane_b32
+; PARTIAL-NOT: v_writelane_b32
+; PARTIAL: .Lfunc_end
+; PARTIAL-LABEL: bitcast_v64f16_to_v128i8_scalar:
+; PARTIAL-COUNT-246: v_writelane_b32
+; PARTIAL-NOT: v_writelane_b32
+; PARTIAL: .Lfunc_end
+; PARTIAL-LABEL: bitcast_v64i16_to_v128i8_scalar:
+; PARTIAL-COUNT-297: v_writelane_b32
+; PARTIAL-NOT: v_writelane_b32
+; PARTIAL: .Lfunc_end
+
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
; FIXME: Currently block machineinstr verifier due to SI BUNDLE pass break physical register liveness. Should remove when the issue is fixed up
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
index 59725b5b5dea6..84f11be32b81a 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
@@ -1,3 +1,20 @@
+; RUN: llc -mtriple=amdgpu6.00 -enable-partial-spills -verify-machineinstrs -verify-regalloc < %s | FileCheck %s --check-prefix=PARTIAL
+
+; Partial scalar-pair spills omit dead high words. Bound the total lane writes
+; in each changed function while retaining the default instruction checks.
+; PARTIAL-LABEL: bitcast_v16f32_to_v64i8_scalar:
+; PARTIAL-COUNT-32: v_writelane_b32
+; PARTIAL-NOT: v_writelane_b32
+; PARTIAL: .Lfunc_end
+; PARTIAL-LABEL: bitcast_v32i16_to_v64i8_scalar:
+; PARTIAL-COUNT-34: v_writelane_b32
+; PARTIAL-NOT: v_writelane_b32
+; PARTIAL: .Lfunc_end
+; PARTIAL-LABEL: bitcast_v32f16_to_v64i8_scalar:
+; PARTIAL-COUNT-48: v_writelane_b32
+; PARTIAL-NOT: v_writelane_b32
+; PARTIAL: .Lfunc_end
+
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
; RUN: llc -mtriple=amdgpu6.00 < %s | FileCheck -check-prefix=SI %s
diff --git a/llvm/test/CodeGen/AMDGPU/dead_bundle.mir b/llvm/test/CodeGen/AMDGPU/dead_bundle.mir
index 77d4a437d167d..d1b5b0702cfbf 100644
--- a/llvm/test/CodeGen/AMDGPU/dead_bundle.mir
+++ b/llvm/test/CodeGen/AMDGPU/dead_bundle.mir
@@ -1,3 +1,4 @@
+# RUN: llc -mtriple=amdgpu11.00--amdpal -enable-partial-spills -verify-machineinstrs -verify-regalloc -start-before=greedy,0 -stop-after=virtregrewriter,0 -stress-regalloc=5 %s -o - | FileCheck %s
# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py
# RUN: llc -mtriple=amdgpu11.00--amdpal -verify-machineinstrs=1 -start-before=greedy,0 -stop-after=virtregrewriter,0 -stress-regalloc=5 %s -o - | FileCheck %s
diff --git a/llvm/test/CodeGen/AMDGPU/greedy-alloc-fail-sgpr1024-spill.mir b/llvm/test/CodeGen/AMDGPU/greedy-alloc-fail-sgpr1024-spill.mir
index 94c22b1aa8664..7459f26134d33 100644
--- a/llvm/test/CodeGen/AMDGPU/greedy-alloc-fail-sgpr1024-spill.mir
+++ b/llvm/test/CodeGen/AMDGPU/greedy-alloc-fail-sgpr1024-spill.mir
@@ -1,3 +1,32 @@
+# RUN: llc -mtriple=amdgpu9.08-amd-amdhsa -start-before=greedy,0 -stop-after=virtregrewriter,0 -enable-partial-spills -verify-machineinstrs -verify-regalloc -o - %s | FileCheck %s --check-prefix=PARTIAL
+
+# The original tuple defines words 0 through 20. Save those words before the
+# clobber and retain the full reload needed by the later tuple consumers.
+# PARTIAL-LABEL: name: greedy_fail_alloc_sgpr1024_spill
+# PARTIAL: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr68, [[SLOT:%stack.[0-9]+]], 0,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr69, [[SLOT]], 4,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr70, [[SLOT]], 8,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr71, [[SLOT]], 12,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr72, [[SLOT]], 16,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr73, [[SLOT]], 20,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr74, [[SLOT]], 24,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr75, [[SLOT]], 28,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr76, [[SLOT]], 32,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr77, [[SLOT]], 36,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr78, [[SLOT]], 40,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr79, [[SLOT]], 44,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr80, [[SLOT]], 48,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr81, [[SLOT]], 52,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr82, [[SLOT]], 56,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr83, [[SLOT]], 60,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr84, [[SLOT]], 64,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr85, [[SLOT]], 68,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr86, [[SLOT]], 72,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr87, [[SLOT]], 76,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial {{(killed )?}}renamable $sgpr88, [[SLOT]], 80,
+# PARTIAL-NOT: SI_SPILL_S32_SAVE_partial
+# PARTIAL: SI_SPILL_S1024_RESTORE [[SLOT]],
+
# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py
# RUN: llc -mtriple=amdgpu9.08-amd-amdhsa -start-before=greedy,0 -stop-after=virtregrewriter,0 -o - %s | FileCheck %s
diff --git a/llvm/test/CodeGen/AMDGPU/infloop-subrange-spill-inspect-subrange.mir b/llvm/test/CodeGen/AMDGPU/infloop-subrange-spill-inspect-subrange.mir
index 63ee4473a56e5..c96943749eb5f 100644
--- a/llvm/test/CodeGen/AMDGPU/infloop-subrange-spill-inspect-subrange.mir
+++ b/llvm/test/CodeGen/AMDGPU/infloop-subrange-spill-inspect-subrange.mir
@@ -1,3 +1,33 @@
+# RUN: llc -enable-partial-spills -verify-machineinstrs -verify-regalloc -mtriple=amdgpu9.00-amd-amdhsa -start-before=greedy,0 -stop-after=virtregrewriter,0 -simplify-mir -o - %s | FileCheck %s --check-prefix=PARTIAL
+
+# Check the spill path through bb.4 without duplicating the full default output.
+# Only the low half is needed at its successor bb.2. The high-half consumer in
+# bb.7 runs before this path and cannot be reached again from bb.4.
+# PARTIAL-LABEL: name: main
+# PARTIAL: bb.0:
+# PARTIAL: renamable $sgpr36_sgpr37_sgpr38_sgpr39_sgpr40_sgpr41_sgpr42_sgpr43_sgpr44_sgpr45_sgpr46_sgpr47_sgpr48_sgpr49_sgpr50_sgpr51 = S_LOAD_DWORDX16_IMM
+# PARTIAL: bb.2:
+# PARTIAL: IMAGE_SAMPLE_LZ_V1_V2 undef {{%[0-9]+}}, killed renamable $sgpr36_sgpr37_sgpr38_sgpr39_sgpr40_sgpr41_sgpr42_sgpr43,
+# PARTIAL: bb.4:
+# PARTIAL: renamable $sgpr12 = IMPLICIT_DEF
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial renamable $sgpr36, [[SLOT:%stack.[0-9]+]], 0, implicit $exec, implicit $sgpr32 :: (store (s32) into [[SLOT]], addrspace 5)
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial renamable $sgpr37, [[SLOT]], 4, implicit $exec, implicit $sgpr32 :: (store (s32) into [[SLOT]] + 4, addrspace 5)
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial renamable $sgpr38, [[SLOT]], 8, implicit $exec, implicit $sgpr32 :: (store (s32) into [[SLOT]] + 8, addrspace 5)
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial renamable $sgpr39, [[SLOT]], 12, implicit $exec, implicit $sgpr32 :: (store (s32) into [[SLOT]] + 12, addrspace 5)
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial renamable $sgpr40, [[SLOT]], 16, implicit $exec, implicit $sgpr32 :: (store (s32) into [[SLOT]] + 16, addrspace 5)
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial renamable $sgpr41, [[SLOT]], 20, implicit $exec, implicit $sgpr32 :: (store (s32) into [[SLOT]] + 20, addrspace 5)
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial renamable $sgpr42, [[SLOT]], 24, implicit $exec, implicit $sgpr32 :: (store (s32) into [[SLOT]] + 24, addrspace 5)
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial killed renamable $sgpr43, [[SLOT]], 28, implicit $exec, implicit $sgpr32 :: (store (s32) into [[SLOT]] + 28, addrspace 5)
+# PARTIAL-NEXT: renamable $sgpr36_sgpr37_sgpr38_sgpr39_sgpr40_sgpr41_sgpr42_sgpr43_sgpr44_sgpr45_sgpr46_sgpr47_sgpr48_sgpr49_sgpr50_sgpr51 = IMPLICIT_DEF
+# PARTIAL-NEXT: dead undef {{%[0-9]+}}.sub0:vreg_96 = IMAGE_SAMPLE_LZ_V1_V2 undef {{%[0-9]+}}, killed renamable $sgpr36_sgpr37_sgpr38_sgpr39_sgpr40_sgpr41_sgpr42_sgpr43,
+# PARTIAL-NEXT: renamable $sgpr36_sgpr37_sgpr38_sgpr39_sgpr40_sgpr41_sgpr42_sgpr43_sgpr44_sgpr45_sgpr46_sgpr47_sgpr48_sgpr49_sgpr50_sgpr51 = SI_SPILL_S512_RESTORE [[SLOT]],
+# PARTIAL-NEXT: {{.*}} = IMPLICIT_DEF
+# PARTIAL-NEXT: dead undef {{%[0-9]+}}.sub0:vreg_128 = IMAGE_SAMPLE_LZ_V1_V2 undef {{%[0-9]+}}, undef renamable $sgpr44_sgpr45_sgpr46_sgpr47_sgpr48_sgpr49_sgpr50_sgpr51,
+# PARTIAL-NEXT: S_BRANCH %bb.2
+# PARTIAL: bb.7:
+# PARTIAL: IMAGE_SAMPLE_LZ_V1_V2 undef {{%[0-9]+}}, renamable $sgpr44_sgpr45_sgpr46_sgpr47_sgpr48_sgpr49_sgpr50_sgpr51,
+# PARTIAL: S_CBRANCH_VCCNZ %bb.7,
+# PARTIAL-NEXT: S_BRANCH %bb.6
# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 4
# RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -verify-regalloc -start-before=greedy,0 -stop-after=virtregrewriter,0 -simplify-mir -o - %s | FileCheck %s
diff --git a/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-buffer-resource.ll b/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-buffer-resource.ll
new file mode 100644
index 0000000000000..019728c7d119b
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-buffer-resource.ll
@@ -0,0 +1,57 @@
+; NOTE: Do not autogenerate. Checks relate spill-slot identities across
+; full-width and partial spill modes.
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -stress-regalloc=2 -stop-after=greedy -verify-machineinstrs -verify-regalloc < %s | FileCheck %s --check-prefixes=CHECK,FULL --implicit-check-not=SI_SPILL_S32_SAVE_partial
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -stress-regalloc=2 -stop-after=greedy -enable-partial-spills -verify-machineinstrs -verify-regalloc < %s | FileCheck %s --check-prefixes=CHECK,PARTIAL
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -stress-regalloc=2 -enable-partial-spills -verify-machineinstrs -verify-regalloc -filetype=null < %s
+
+; A buffer resource requires four SGPR words, but its fields can be defined at
+; different times. Greedy splits the partially constructed resource under
+; pressure. Save each available word at its original offset, then complete the
+; resource before the buffer load. A fully constructed resource still spills
+; at full width. The LLVM IR supplies every field and no split metadata.
+
+; CHECK-LABEL: name: spill_partial_buffer_resource
+; CHECK: body:
+; CHECK: [[FIELDS:%[0-9]+]].sub3:sgpr_128 = S_MOV_B32 131072
+; CHECK-NEXT: [[FIELDS]].sub2:sgpr_128 = S_MOV_B32 1024
+; FULL-NEXT: SI_SPILL_S128_SAVE [[FIELDS]], [[SLOT:%stack\.[0-9]+]],
+; PARTIAL-NEXT: [[PART:%[0-9]+]].sub2_sub3:sgpr_128 = lr-split COPY [[FIELDS]].sub2_sub3
+; PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[PART]].sub2, [[SLOT:%stack\.[0-9]+]], 8,
+; PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[PART]].sub3, [[SLOT]], 12,
+; CHECK: [[RELOAD:%[0-9]+]]:sgpr_128 = SI_SPILL_S128_RESTORE [[SLOT]],
+; CHECK-NEXT: [[UPPER:%[0-9]+]].sub2_sub3:sgpr_128 = lr-split COPY [[RELOAD]].sub2_sub3
+; CHECK-NEXT: [[UPPER]].sub1:sgpr_128 = S_AND_B32
+; CHECK-NEXT: [[PART2:%[0-9]+]].sub2_sub3:sgpr_128 = lr-split COPY [[UPPER]].sub2_sub3 {
+; CHECK-NEXT: internal [[PART2]].sub1:sgpr_128 = lr-split COPY [[UPPER]].sub1
+; CHECK-NEXT: }
+; FULL-NEXT: SI_SPILL_S128_SAVE [[PART2]], [[SLOT]],
+; PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[PART2]].sub1, [[SLOT]], 4,
+; PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[PART2]].sub2, [[SLOT]], 8,
+; PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[PART2]].sub3, [[SLOT]], 12,
+; CHECK: [[COMPLETE:%[0-9]+]]:sgpr_128 = SI_SPILL_S128_RESTORE [[SLOT]],
+; CHECK-NEXT: [[COMPLETE]].sub0:sgpr_128 = COPY
+; CHECK-NEXT: SI_SPILL_S128_SAVE [[COMPLETE]], [[SLOT]],
+; CHECK-NEXT: [[RESOURCE:%[0-9]+]]:sgpr_128 = SI_SPILL_S128_RESTORE [[SLOT]],
+; CHECK: BUFFER_LOAD_DWORD_OFFEN {{%[0-9]+}}, [[RESOURCE]],
+; CHECK: S_ENDPGM
+
+define amdgpu_kernel void @spill_partial_buffer_resource(ptr addrspace(1) %in, ptr addrspace(1) %out, i32 %count) {
+entry:
+ %tid = call i32 @llvm.amdgcn.workitem.id.x()
+ %offset = shl i32 %tid, 2
+ %resource = call ptr addrspace(8) @llvm.amdgcn.make.buffer.rsrc.p8.p1.i64(ptr addrspace(1) %in, i16 0, i64 1024, i32 131072)
+ br label %loop
+
+loop:
+ %i = phi i32 [ 0, %entry ], [ %inc, %loop ]
+ %sum = phi i32 [ 0, %entry ], [ %next, %loop ]
+ %value = call i32 @llvm.amdgcn.raw.ptr.buffer.load.i32(ptr addrspace(8) %resource, i32 %offset, i32 0, i32 0)
+ %next = add i32 %sum, %value
+ store atomic i32 %next, ptr addrspace(1) %out monotonic, align 4
+ %inc = add i32 %i, 1
+ %done = icmp eq i32 %inc, %count
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-cleanup.mir b/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-cleanup.mir
new file mode 100644
index 0000000000000..3b2bc48c11c52
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-cleanup.mir
@@ -0,0 +1,353 @@
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -run-pass=si-lower-sgpr-spills -verify-machineinstrs %s -o - | FileCheck %s --check-prefixes=CHECK,OPT
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx900 -run-pass=liveintervals,si-lower-sgpr-spills -verify-machineinstrs %s -o - | FileCheck %s --check-prefixes=CHECK,OPT
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1100 -mattr=+wavefrontsize32,-wavefrontsize64 -run-pass=liveintervals,si-lower-sgpr-spills -verify-machineinstrs %s -o - | FileCheck %s --check-prefixes=CHECK,OPT
+
+# These narrow checks isolate the optional word store; other lane
+# instructions are shared with ordinary spill-lowering tests.
+# A reload proves the current value of a slot word. Copies retain that
+# proof; changes to the source or slot word invalidate it. Keep the proof
+# inside one block and stop at unknown memory effects.
+# CHECK-LABEL: name: same_word
+# OPT-NOT: SI_SPILL_S32_TO_VGPR killed $sgpr5,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: copied_high_word
+# OPT-NOT: SI_SPILL_S32_TO_VGPR killed $sgpr9,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: other_word_changed
+# OPT-NOT: SI_SPILL_S32_TO_VGPR killed $sgpr5,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: source_changed
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr4, 0,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: source_alias_changed
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr5, 1,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: slot_word_changed
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr5, 1,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: slot_fully_changed
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr5, 1,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: wrong_word
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr4, 2,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: different_slot
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr5, 1,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: unknown_store
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr5, 1,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: volatile_store
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr5, 1,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: volatile_reload
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr5, 1,
+# CHECK: S_ENDPGM
+# CHECK-LABEL: name: cross_block
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr5, 1,
+# CHECK: S_ENDPGM
+--- |
+ define amdgpu_kernel void @same_word() { ret void }
+ define amdgpu_kernel void @copied_high_word() { ret void }
+ define amdgpu_kernel void @other_word_changed() { ret void }
+ define amdgpu_kernel void @source_changed() { ret void }
+ define amdgpu_kernel void @source_alias_changed() { ret void }
+ define amdgpu_kernel void @slot_word_changed() { ret void }
+ define amdgpu_kernel void @slot_fully_changed() { ret void }
+ define amdgpu_kernel void @wrong_word() { ret void }
+ define amdgpu_kernel void @different_slot() { ret void }
+ define amdgpu_kernel void @unknown_store() { ret void }
+ define amdgpu_kernel void @volatile_store() { ret void }
+ define amdgpu_kernel void @volatile_reload() { ret void }
+ define amdgpu_kernel void @cross_block() { ret void }
+...
+---
+name: same_word
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ SI_SPILL_S32_SAVE_partial killed $sgpr5, %stack.0, 4, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: copied_high_word
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr8_sgpr9 = COPY $sgpr6_sgpr7
+ SI_SPILL_S32_SAVE_partial killed $sgpr9, %stack.0, 12, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: other_word_changed
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr20 = S_MOV_B32 101
+ SI_SPILL_S32_SAVE_partial killed $sgpr20, %stack.0, 12, implicit $exec, implicit $sgpr32
+ SI_SPILL_S32_SAVE_partial killed $sgpr5, %stack.0, 4, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: source_changed
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4 = S_ADD_U32 $sgpr4, 1, implicit-def dead $scc
+ SI_SPILL_S32_SAVE_partial killed $sgpr4, %stack.0, 0, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: source_alias_changed
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5 = S_MOV_B64 0
+ SI_SPILL_S32_SAVE_partial killed $sgpr5, %stack.0, 4, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: slot_word_changed
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr20 = S_MOV_B32 101
+ SI_SPILL_S32_SAVE_partial killed $sgpr20, %stack.0, 4, implicit $exec, implicit $sgpr32
+ SI_SPILL_S32_SAVE_partial killed $sgpr5, %stack.0, 4, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: slot_fully_changed
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr17 = S_MOV_B32 101
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ SI_SPILL_S32_SAVE_partial killed $sgpr5, %stack.0, 4, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: wrong_word
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ SI_SPILL_S32_SAVE_partial killed $sgpr4, %stack.0, 8, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: different_slot
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+ - { id: 1, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr17 = S_MOV_B32 201
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.1, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.1, implicit $exec, implicit $sgpr32
+ SI_SPILL_S32_SAVE_partial killed $sgpr5, %stack.0, 4, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: unknown_store
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ INLINEASM &"", 25
+ SI_SPILL_S32_SAVE_partial killed $sgpr5, %stack.0, 4, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: volatile_store
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ SI_SPILL_S32_SAVE_partial killed $sgpr5, %stack.0, 4, implicit $exec, implicit $sgpr32 :: (volatile store (s32) into %stack.0 + 4, addrspace 5)
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: volatile_reload
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32 :: (volatile load (s128) from %stack.0, align 4, addrspace 5)
+ SI_SPILL_S32_SAVE_partial killed $sgpr5, %stack.0, 4, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: cross_block
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr4_sgpr5_sgpr6_sgpr7 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_BRANCH %bb.1
+
+ bb.1:
+ liveins: $sgpr5
+ SI_SPILL_S32_SAVE_partial killed $sgpr5, %stack.0, 4, implicit $exec, implicit $sgpr32
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
diff --git a/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-loaded-tuple.ll b/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-loaded-tuple.ll
new file mode 100644
index 0000000000000..d1cc6978ebc57
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-loaded-tuple.ll
@@ -0,0 +1,56 @@
+; NOTE: Do not autogenerate. Checks relate spill-slot identities across
+; full-width and partial spill modes.
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -stress-regalloc=2 -stop-after=greedy -verify-machineinstrs -verify-regalloc < %s | FileCheck %s --check-prefixes=CHECK,FULL --implicit-check-not=SI_SPILL_S32_SAVE_partial
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -stress-regalloc=2 -stop-after=greedy -enable-partial-spills -verify-machineinstrs -verify-regalloc < %s | FileCheck %s --check-prefixes=CHECK,PARTIAL
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -stress-regalloc=2 -enable-partial-spills -verify-machineinstrs -verify-regalloc -filetype=null < %s
+
+; Like a wide load of scalar arguments, each tuple has words used early and a
+; word needed throughout a later loop. Volatile accesses keep those uses ordered.
+; Limit registers so greedy creates the partial split from ordinary LLVM IR.
+; Only word 2 of the second tuple needs saving after its early uses.
+
+; CHECK-LABEL: name: spill_loaded_tuple_word
+; CHECK: body:
+; CHECK: [[ARGS:%[0-9]+]]:sgpr_128 = S_LOAD_DWORDX4_IMM {{.*}}, 16, 0 :: (volatile load
+; FULL-NEXT: SI_SPILL_S128_SAVE [[ARGS]], [[SLOT:%stack\.[0-9]+]],
+; CHECK: S_MUL_I32 [[ARGS]].sub0, [[ARGS]].sub1
+; PARTIAL-NEXT: [[PART:%[0-9]+]].sub2:sgpr_128 = lr-split COPY [[ARGS]].sub2
+; PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[PART]].sub2, [[SLOT:%stack\.[0-9]+]], 8,
+; PARTIAL-NOT: SI_SPILL_S128_SAVE {{.*}}, [[SLOT]],
+; CHECK: [[RELOAD:%[0-9]+]]:sgpr_128 = SI_SPILL_S128_RESTORE [[SLOT]],
+; CHECK-NEXT: [[LATE:%[0-9]+]].sub2:sgpr_128 = lr-split COPY [[RELOAD]].sub2
+; CHECK: S_XOR_B32 {{%[0-9]+}}, [[LATE]].sub2,
+; CHECK: S_ENDPGM
+
+define amdgpu_kernel void @spill_loaded_tuple_word(ptr addrspace(1) %out, ptr addrspace(4) %input, i32 %count) {
+entry:
+ %tid = call i32 @llvm.amdgcn.workitem.id.x()
+ %out.ptr = getelementptr i32, ptr addrspace(1) %out, i32 %tid
+ %first = load volatile <4 x i32>, ptr addrspace(4) %input, align 16
+ %first.a = extractelement <4 x i32> %first, i32 0
+ %first.b = extractelement <4 x i32> %first, i32 1
+ %first.late = extractelement <4 x i32> %first, i32 2
+ %first.product = mul i32 %first.a, %first.b
+ store volatile i32 %first.product, ptr addrspace(1) %out.ptr
+ %second.ptr = getelementptr <4 x i32>, ptr addrspace(4) %input, i32 1
+ %second = load volatile <4 x i32>, ptr addrspace(4) %second.ptr, align 16
+ %second.a = extractelement <4 x i32> %second, i32 0
+ %second.b = extractelement <4 x i32> %second, i32 1
+ %second.late = extractelement <4 x i32> %second, i32 2
+ %second.product = mul i32 %second.a, %second.b
+ store volatile i32 %second.product, ptr addrspace(1) %out.ptr
+ br label %loop
+
+loop:
+ %i = phi i32 [ 0, %entry ], [ %inc, %loop ]
+ %sum = phi i32 [ 0, %entry ], [ %next, %loop ]
+ %product = mul i32 %sum, %first.late
+ %next = xor i32 %product, %second.late
+ store volatile i32 %next, ptr addrspace(1) %out.ptr
+ %inc = add i32 %i, 1
+ %done = icmp eq i32 %inc, %count
+ br i1 %done, label %exit, label %loop
+
+exit:
+ ret void
+}
diff --git a/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-offset.mir b/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-offset.mir
new file mode 100644
index 0000000000000..5c220ca54d435
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-offset.mir
@@ -0,0 +1,169 @@
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx900 -run-pass=liveintervals,si-lower-sgpr-spills -verify-machineinstrs %s -o - | FileCheck %s
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -run-pass=si-lower-sgpr-spills -verify-machineinstrs %s -o - | FileCheck %s
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx1010 -mattr=+wavefrontsize32,-wavefrontsize64 -run-pass=si-lower-sgpr-spills -verify-machineinstrs %s -o - | FileCheck %s
+
+# Each partial store updates its original word in the full-size spill slot.
+# The last case places another slot first, so the allocated VGPR lane differs
+# from the word index inside the slot.
+# CHECK-LABEL: name: low
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr20, 0,
+# CHECK-LABEL: name: high
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr20, 3,
+# CHECK-LABEL: name: sparse
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr20, 0,
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr20, 2,
+# CHECK-LABEL: name: repeat
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr20, 2,
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr20, 2,
+# CHECK-LABEL: name: packed_offset
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr20, 7,
+# CHECK-LABEL: name: cross_block
+# CHECK: SI_SPILL_S32_TO_VGPR killed $sgpr20, 3,
+--- |
+ define amdgpu_kernel void @low() { ret void }
+ define amdgpu_kernel void @high() { ret void }
+ define amdgpu_kernel void @sparse() { ret void }
+ define amdgpu_kernel void @repeat() { ret void }
+ define amdgpu_kernel void @packed_offset() { ret void }
+ define amdgpu_kernel void @cross_block() { ret void }
+...
+---
+name: low
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+ - { id: 1, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr20 = S_MOV_B32 101
+ SI_SPILL_S32_SAVE_partial killed $sgpr20, %stack.0, 0, implicit $exec, implicit $sgpr32 :: (store (s32) into %stack.0 + 0)
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: high
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+ - { id: 1, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr20 = S_MOV_B32 104
+ SI_SPILL_S32_SAVE_partial killed $sgpr20, %stack.0, 12, implicit $exec, implicit $sgpr32 :: (store (s32) into %stack.0 + 12)
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: sparse
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+ - { id: 1, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr20 = S_MOV_B32 101
+ SI_SPILL_S32_SAVE_partial killed $sgpr20, %stack.0, 0, implicit $exec, implicit $sgpr32 :: (store (s32) into %stack.0 + 0)
+ $sgpr20 = S_MOV_B32 103
+ SI_SPILL_S32_SAVE_partial killed $sgpr20, %stack.0, 8, implicit $exec, implicit $sgpr32 :: (store (s32) into %stack.0 + 8)
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: repeat
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+ - { id: 1, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr20 = S_MOV_B32 102
+ SI_SPILL_S32_SAVE_partial killed $sgpr20, %stack.0, 8, implicit $exec, implicit $sgpr32 :: (store (s32) into %stack.0 + 8)
+ $sgpr20 = S_MOV_B32 103
+ SI_SPILL_S32_SAVE_partial killed $sgpr20, %stack.0, 8, implicit $exec, implicit $sgpr32 :: (store (s32) into %stack.0 + 8)
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: packed_offset
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+ - { id: 1, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.1, implicit $exec, implicit $sgpr32
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr20 = S_MOV_B32 104
+ SI_SPILL_S32_SAVE_partial killed $sgpr20, %stack.0, 12, implicit $exec, implicit $sgpr32 :: (store (s32) into %stack.0 + 12)
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
+---
+name: cross_block
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+ - { id: 1, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ $sgpr16 = S_MOV_B32 11
+ $sgpr17 = S_MOV_B32 21
+ $sgpr18 = S_MOV_B32 31
+ $sgpr19 = S_MOV_B32 41
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ S_BRANCH %bb.1
+
+ bb.1:
+ $sgpr20 = S_MOV_B32 104
+ SI_SPILL_S32_SAVE_partial killed $sgpr20, %stack.0, 12, implicit $exec, implicit $sgpr32 :: (store (s32) into %stack.0 + 12)
+ $sgpr24_sgpr25_sgpr26_sgpr27 = SI_SPILL_S128_RESTORE %stack.0, implicit $exec, implicit $sgpr32
+ S_ENDPGM 0, implicit $sgpr24_sgpr25_sgpr26_sgpr27
+...
diff --git a/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-scavenge.mir b/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-scavenge.mir
new file mode 100644
index 0000000000000..9acf178bff01c
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/partial-sgpr-spill-scavenge.mir
@@ -0,0 +1,90 @@
+# RUN: llc -mtriple=amdgcn-amd-unknown -mcpu=gfx900 -amdgpu-spill-sgpr-to-vgpr=false -start-before=si-lower-sgpr-spills -stop-after=prolog-epilog -verify-machineinstrs %s -o - | FileCheck %s --check-prefixes=CHECK,W64
+# RUN: llc -mtriple=amdgcn-amd-unknown -mcpu=gfx1010 -mattr=+wavefrontsize32,-wavefrontsize64 -amdgpu-spill-sgpr-to-vgpr=false -start-before=si-lower-sgpr-spills -stop-after=prolog-epilog -verify-machineinstrs %s -o - | FileCheck %s --check-prefixes=CHECK,W32
+
+# A partial spill must preserve the other words when no SGPR can save EXEC.
+# Keep the allocatable SGPRs (including VCC_HI in wave32) live across the spill.
+# The second function exhausts the VGPRs too, requiring both active and inactive
+# lanes to be restored.
+# Hand-written checks focus on this preservation sequence and omit the long
+# register-pressure operands.
+
+--- |
+ define amdgpu_kernel void @no_sgpr() #0 { ret void }
+ define amdgpu_kernel void @no_sgpr_or_vgpr() #1 { ret void }
+ attributes #0 = { "amdgpu-num-sgpr"="40" }
+ attributes #1 = { "amdgpu-num-sgpr"="40" "amdgpu-num-vgpr"="1" }
+...
+---
+name: no_sgpr
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ scratchRSrcReg: '$sgpr96_sgpr97_sgpr98_sgpr99'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ liveins: $vcc, $vcc_hi, $sgpr0, $sgpr1, $sgpr2, $sgpr3, $sgpr4, $sgpr5, $sgpr6, $sgpr7, $sgpr8, $sgpr9, $sgpr10, $sgpr11, $sgpr12, $sgpr13, $sgpr14, $sgpr15, $sgpr16, $sgpr17, $sgpr18, $sgpr19, $sgpr20, $sgpr21, $sgpr22, $sgpr23, $sgpr24, $sgpr25, $sgpr26, $sgpr27, $sgpr28, $sgpr29, $sgpr30, $sgpr31, $sgpr32, $sgpr33, $sgpr34, $sgpr35, $sgpr36, $sgpr37, $sgpr38, $sgpr39
+ ; CHECK-LABEL: name: no_sgpr
+ ; CHECK: $sgpr20 = S_MOV_B32 104
+ ; W64-NEXT: [[EXEC:\$exec]] = [[NOT:S_NOT_B64]] [[EXEC]], implicit-def dead $scc, implicit-def $vgpr0
+ ; W32-NEXT: [[EXEC:\$exec_lo]] = [[NOT:S_NOT_B32]] [[EXEC]], implicit-def dead $scc, implicit-def $vgpr0
+ ; CHECK-NEXT: BUFFER_STORE_DWORD_OFFSET killed $vgpr0, $sgpr96_sgpr97_sgpr98_sgpr99, 0, 16, 0, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr0 = BUFFER_LOAD_DWORD_OFFSET $sgpr96_sgpr97_sgpr98_sgpr99, 0, 0, 0, 0, implicit $exec
+ ; CHECK-NEXT: [[EXEC]] = [[NOT]] [[EXEC]], implicit-def dead $scc
+ ; CHECK-NEXT: $vgpr0 = BUFFER_LOAD_DWORD_OFFSET $sgpr96_sgpr97_sgpr98_sgpr99, 0, 0, 0, 0, implicit $exec
+ ; CHECK-NEXT: [[EXEC]] = [[NOT]] [[EXEC]], implicit-def dead $scc
+ ; CHECK-NEXT: $vgpr0 = SI_SPILL_S32_TO_VGPR $sgpr20, 3, $vgpr0
+ ; CHECK-NEXT: BUFFER_STORE_DWORD_OFFSET $vgpr0, $sgpr96_sgpr97_sgpr98_sgpr99, 0, 0, 0, 0, implicit $exec
+ ; CHECK-NEXT: [[EXEC]] = [[NOT]] [[EXEC]], implicit-def dead $scc
+ ; CHECK-NEXT: BUFFER_STORE_DWORD_OFFSET killed $vgpr0, $sgpr96_sgpr97_sgpr98_sgpr99, 0, 0, 0, 0, implicit $exec
+ ; CHECK-NEXT: [[EXEC]] = [[NOT]] [[EXEC]], implicit-def dead $scc
+ ; CHECK-NEXT: $vgpr0 = BUFFER_LOAD_DWORD_OFFSET $sgpr96_sgpr97_sgpr98_sgpr99, 0, 16, 0, 0, implicit $exec
+ ; CHECK-NEXT: [[EXEC]] = [[NOT]] [[EXEC]], implicit-def dead $scc, implicit killed $vgpr0
+ ; CHECK-NEXT: S_ENDPGM 0
+
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr20 = S_MOV_B32 104
+ SI_SPILL_S32_SAVE_partial $sgpr20, %stack.0, 12, implicit $exec, implicit $sgpr32 :: (store (s32) into %stack.0 + 12)
+ S_ENDPGM 0, implicit $vcc, implicit $vcc_hi, implicit $sgpr0, implicit $sgpr1, implicit $sgpr2, implicit $sgpr3, implicit $sgpr4, implicit $sgpr5, implicit $sgpr6, implicit $sgpr7, implicit $sgpr8, implicit $sgpr9, implicit $sgpr10, implicit $sgpr11, implicit $sgpr12, implicit $sgpr13, implicit $sgpr14, implicit $sgpr15, implicit $sgpr16, implicit $sgpr17, implicit $sgpr18, implicit $sgpr19, implicit $sgpr20, implicit $sgpr21, implicit $sgpr22, implicit $sgpr23, implicit $sgpr24, implicit $sgpr25, implicit $sgpr26, implicit $sgpr27, implicit $sgpr28, implicit $sgpr29, implicit $sgpr30, implicit $sgpr31, implicit $sgpr32, implicit $sgpr33, implicit $sgpr34, implicit $sgpr35, implicit $sgpr36, implicit $sgpr37, implicit $sgpr38, implicit $sgpr39
+...
+---
+name: no_sgpr_or_vgpr
+tracksRegLiveness: true
+stack:
+ - { id: 0, type: spill-slot, size: 16, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ stackPtrOffsetReg: '$sgpr32'
+ scratchRSrcReg: '$sgpr96_sgpr97_sgpr98_sgpr99'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ liveins: $vcc, $vcc_hi, $sgpr0, $sgpr1, $sgpr2, $sgpr3, $sgpr4, $sgpr5, $sgpr6, $sgpr7, $sgpr8, $sgpr9, $sgpr10, $sgpr11, $sgpr12, $sgpr13, $sgpr14, $sgpr15, $sgpr16, $sgpr17, $sgpr18, $sgpr19, $sgpr20, $sgpr21, $sgpr22, $sgpr23, $sgpr24, $sgpr25, $sgpr26, $sgpr27, $sgpr28, $sgpr29, $sgpr30, $sgpr31, $sgpr32, $sgpr33, $sgpr34, $sgpr35, $sgpr36, $sgpr37, $sgpr38, $sgpr39, $vgpr0
+ ; CHECK-LABEL: name: no_sgpr_or_vgpr
+ ; CHECK: $sgpr20 = S_MOV_B32 104
+ ; CHECK-NEXT: BUFFER_STORE_DWORD_OFFSET $vgpr0, $sgpr96_sgpr97_sgpr98_sgpr99, 0, 16, 0, 0, implicit $exec
+ ; W64-NEXT: [[EXEC:\$exec]] = [[NOT:S_NOT_B64]] [[EXEC]], implicit-def dead $scc
+ ; W32-NEXT: [[EXEC:\$exec_lo]] = [[NOT:S_NOT_B32]] [[EXEC]], implicit-def dead $scc
+ ; CHECK-NEXT: BUFFER_STORE_DWORD_OFFSET killed $vgpr0, $sgpr96_sgpr97_sgpr98_sgpr99, 0, 16, 0, 0, implicit $exec
+ ; CHECK-NEXT: $vgpr0 = BUFFER_LOAD_DWORD_OFFSET $sgpr96_sgpr97_sgpr98_sgpr99, 0, 0, 0, 0, implicit $exec
+ ; CHECK-NEXT: [[EXEC]] = [[NOT]] [[EXEC]], implicit-def dead $scc
+ ; CHECK-NEXT: $vgpr0 = BUFFER_LOAD_DWORD_OFFSET $sgpr96_sgpr97_sgpr98_sgpr99, 0, 0, 0, 0, implicit $exec
+ ; CHECK-NEXT: [[EXEC]] = [[NOT]] [[EXEC]], implicit-def dead $scc
+ ; CHECK-NEXT: $vgpr0 = SI_SPILL_S32_TO_VGPR $sgpr20, 3, $vgpr0
+ ; CHECK-NEXT: BUFFER_STORE_DWORD_OFFSET $vgpr0, $sgpr96_sgpr97_sgpr98_sgpr99, 0, 0, 0, 0, implicit $exec
+ ; CHECK-NEXT: [[EXEC]] = [[NOT]] [[EXEC]], implicit-def dead $scc
+ ; CHECK-NEXT: BUFFER_STORE_DWORD_OFFSET killed $vgpr0, $sgpr96_sgpr97_sgpr98_sgpr99, 0, 0, 0, 0, implicit $exec
+ ; CHECK-NEXT: [[EXEC]] = [[NOT]] [[EXEC]], implicit-def dead $scc
+ ; CHECK-NEXT: $vgpr0 = BUFFER_LOAD_DWORD_OFFSET $sgpr96_sgpr97_sgpr98_sgpr99, 0, 16, 0, 0, implicit $exec
+ ; CHECK-NEXT: [[EXEC]] = [[NOT]] [[EXEC]], implicit-def dead $scc
+ ; CHECK-NEXT: $vgpr0 = BUFFER_LOAD_DWORD_OFFSET $sgpr96_sgpr97_sgpr98_sgpr99, 0, 16, 0, 0, implicit $exec
+ ; CHECK-NEXT: S_ENDPGM 0
+
+ SI_SPILL_S128_SAVE $sgpr16_sgpr17_sgpr18_sgpr19, %stack.0, implicit $exec, implicit $sgpr32
+ $sgpr20 = S_MOV_B32 104
+ SI_SPILL_S32_SAVE_partial $sgpr20, %stack.0, 12, implicit $exec, implicit $sgpr32 :: (store (s32) into %stack.0 + 12)
+ S_ENDPGM 0, implicit $vcc, implicit $vcc_hi, implicit $sgpr0, implicit $sgpr1, implicit $sgpr2, implicit $sgpr3, implicit $sgpr4, implicit $sgpr5, implicit $sgpr6, implicit $sgpr7, implicit $sgpr8, implicit $sgpr9, implicit $sgpr10, implicit $sgpr11, implicit $sgpr12, implicit $sgpr13, implicit $sgpr14, implicit $sgpr15, implicit $sgpr16, implicit $sgpr17, implicit $sgpr18, implicit $sgpr19, implicit $sgpr20, implicit $sgpr21, implicit $sgpr22, implicit $sgpr23, implicit $sgpr24, implicit $sgpr25, implicit $sgpr26, implicit $sgpr27, implicit $sgpr28, implicit $sgpr29, implicit $sgpr30, implicit $sgpr31, implicit $sgpr32, implicit $sgpr33, implicit $sgpr34, implicit $sgpr35, implicit $sgpr36, implicit $sgpr37, implicit $sgpr38, implicit $sgpr39, implicit $vgpr0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/ra-inserted-scalar-instructions.mir b/llvm/test/CodeGen/AMDGPU/ra-inserted-scalar-instructions.mir
index c670df94f616e..e7ad936396c02 100644
--- a/llvm/test/CodeGen/AMDGPU/ra-inserted-scalar-instructions.mir
+++ b/llvm/test/CodeGen/AMDGPU/ra-inserted-scalar-instructions.mir
@@ -1,3 +1,24 @@
+# RUN: llc -enable-partial-spills -verify-regalloc -mtriple=amdgpu10.30-amd-amdpal -run-pass=greedy --stress-regalloc=6 --verify-machineinstrs -o - %s | FileCheck %s --check-prefix=PARTIAL
+# Check the partial store and its consumer separately from the full default
+# output. The high word comes from the live-in; the low word is defined later.
+# PARTIAL-LABEL: name: test_kernel
+# PARTIAL: bb.0:
+# PARTIAL: undef [[IN:%[0-9]+]].sub1:sgpr_64 = COPY $sgpr0
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[IN]].sub1, [[SLOT:%stack.[0-9]+]], 4, implicit $exec, implicit $sgpr32 :: (store (s32) into [[SLOT]] + 4, addrspace 5)
+# PARTIAL: bb.6:
+# PARTIAL: [[RELOAD:%[0-9]+]]:sgpr_64 = SI_SPILL_S64_RESTORE [[SLOT]],
+# PARTIAL-NEXT: undef [[VALUE:%[0-9]+]].sub1:sgpr_64 = lr-split COPY [[RELOAD]].sub1
+# PARTIAL-NEXT: [[VALUE]].sub0:sgpr_64 = S_MOV_B32 1
+# PARTIAL: bb.7:
+# PARTIAL: SI_SPILL_S64_SAVE [[VALUE]], [[SLOT]],
+# PARTIAL: bb.9:
+# PARTIAL-NEXT: successors: %bb.10{{.*}}
+# PARTIAL-NEXT: {{^ +$}}
+# PARTIAL-NEXT: {{%[0-9]+}}:sgpr_32 = lr-split COPY {{%[0-9]+}}
+# PARTIAL-NEXT: [[VALUE]]:sgpr_64 = SI_SPILL_S64_RESTORE [[SLOT]],
+# PARTIAL: bb.10:
+# PARTIAL: S_LOAD_DWORD_IMM [[VALUE]], 0, 0
+
# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 4
# RUN: llc -mtriple=amdgpu10.30-amd-amdpal -run-pass=greedy --stress-regalloc=6 --verify-machineinstrs -o - %s | FileCheck -check-prefix=GCN %s
diff --git a/llvm/test/CodeGen/AMDGPU/ran-out-of-sgprs-allocation-failure.mir b/llvm/test/CodeGen/AMDGPU/ran-out-of-sgprs-allocation-failure.mir
index 794d96dfa2365..714e61187218c 100644
--- a/llvm/test/CodeGen/AMDGPU/ran-out-of-sgprs-allocation-failure.mir
+++ b/llvm/test/CodeGen/AMDGPU/ran-out-of-sgprs-allocation-failure.mir
@@ -1,3 +1,22 @@
+# RUN: llc -enable-partial-spills -verify-machineinstrs -verify-regalloc -mtriple=amdgpu9.0a-amd-amdhsa -start-before=greedy,0 -stop-after=virtregrewriter,0 -greedy-regclass-priority-trumps-globalness=1 -o - %s | FileCheck %s --check-prefix=PARTIAL
+# Check the two defined words through their stores, reloads and consumers.
+# The other words of this tuple are undefined.
+# PARTIAL-LABEL: name: need_large_tuple_split
+# PARTIAL: bb.0:
+# PARTIAL: renamable $sgpr52 = S_MOV_B32 0
+# PARTIAL: renamable $sgpr53 = S_MOV_B32 1083786240
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial renamable $sgpr52, [[SLOT:%stack.[0-9]+]], 64, implicit $exec, implicit $sgpr32 :: (store (s32) into [[SLOT]] + 64, addrspace 5)
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial killed renamable $sgpr53, [[SLOT]], 68, implicit $exec, implicit $sgpr32 :: (store (s32) into [[SLOT]] + 68, addrspace 5)
+# PARTIAL-NEXT: S_BRANCH %bb.1
+# PARTIAL: bb.2:
+# PARTIAL: renamable $sgpr36_sgpr37_sgpr38_sgpr39_sgpr40_sgpr41_sgpr42_sgpr43_sgpr44_sgpr45_sgpr46_sgpr47_sgpr48_sgpr49_sgpr50_sgpr51_sgpr52_sgpr53_sgpr54_sgpr55_sgpr56_sgpr57_sgpr58_sgpr59_sgpr60_sgpr61_sgpr62_sgpr63_sgpr64_sgpr65_sgpr66_sgpr67 = SI_SPILL_S1024_RESTORE [[SLOT]],
+# PARTIAL-NEXT: renamable $sgpr68_sgpr69 = lr-split COPY killed renamable $sgpr52_sgpr53
+# PARTIAL-NEXT: renamable $sgpr36 = COPY renamable $sgpr68
+# PARTIAL: bb.9:
+# PARTIAL: renamable $sgpr36_sgpr37_sgpr38_sgpr39_sgpr40_sgpr41_sgpr42_sgpr43_sgpr44_sgpr45_sgpr46_sgpr47_sgpr48_sgpr49_sgpr50_sgpr51_sgpr52_sgpr53_sgpr54_sgpr55_sgpr56_sgpr57_sgpr58_sgpr59_sgpr60_sgpr61_sgpr62_sgpr63_sgpr64_sgpr65_sgpr66_sgpr67 = SI_SPILL_S1024_RESTORE [[SLOT]],
+# PARTIAL-NEXT: [[PAIR:%[0-9]+]]:vreg_64_align2 = COPY killed renamable $sgpr52_sgpr53, implicit $exec
+# PARTIAL-NEXT: GLOBAL_STORE_DWORDX2_SADDR {{.*}}, [[PAIR]],
+
# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py
# RUN: llc -mtriple=amdgpu9.0a-amd-amdhsa -verify-regalloc -start-before=greedy,0 -stop-after=virtregrewriter,0 -greedy-regclass-priority-trumps-globalness=1 -o - %s | FileCheck %s
diff --git a/llvm/test/CodeGen/AMDGPU/spill-scavenge-offset.ll b/llvm/test/CodeGen/AMDGPU/spill-scavenge-offset.ll
index 5c829d7ec88ef..fcc2766700f44 100644
--- a/llvm/test/CodeGen/AMDGPU/spill-scavenge-offset.ll
+++ b/llvm/test/CodeGen/AMDGPU/spill-scavenge-offset.ll
@@ -1,3 +1,26 @@
+; RUN: llc -mtriple=amdgpu6.01 -enable-misched=0 -post-RA-scheduler=0 -amdgpu-spill-sgpr-to-vgpr=0 -enable-partial-spills -verify-machineinstrs -verify-regalloc < %s | FileCheck %s --check-prefix=PARTIAL
+
+; Partial scratch stores select the original word with EXEC. Each store must
+; save and restore the scavenged VGPR and restore the incoming EXEC value.
+; PARTIAL-LABEL: test_limited_sgpr:
+; PARTIAL: s_mov_b64 [[EXEC:s\[[0-9]+:[0-9]+\]]], exec
+; PARTIAL: s_mov_b64 exec, 1
+; PARTIAL: buffer_store_dword [[TMP:v[0-9]+]],
+; PARTIAL: v_writelane_b32 [[TMP]], s0, 0
+; PARTIAL: buffer_store_dword [[TMP]],
+; PARTIAL: buffer_load_dword [[TMP]],
+; PARTIAL: s_waitcnt vmcnt(0)
+; PARTIAL-NEXT: s_mov_b64 exec, [[EXEC]]
+; PARTIAL: s_mov_b64 exec, 2
+; PARTIAL: v_writelane_b32 {{v[0-9]+}}, s1, 1
+; PARTIAL: s_mov_b64 exec, 4
+; PARTIAL: v_writelane_b32 {{v[0-9]+}}, s2, 2
+; PARTIAL: s_mov_b64 exec, 8
+; PARTIAL: v_writelane_b32 {{v[0-9]+}}, s3, 3
+; PARTIAL: buffer_load_dword
+; PARTIAL: s_waitcnt vmcnt(0)
+; PARTIAL-NEXT: s_mov_b64 exec,
+
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
; RUN: llc -mtriple=amdgpu6.01 -enable-misched=0 -post-RA-scheduler=0 -amdgpu-spill-sgpr-to-vgpr=0 < %s | FileCheck -check-prefixes=CHECK,GFX6 %s
; RUN: llc -sgpr-regalloc=basic -vgpr-regalloc=basic -mtriple=amdgpu8.02 -enable-misched=0 -post-RA-scheduler=0 -amdgpu-spill-sgpr-to-vgpr=0 < %s | FileCheck --check-prefix=CHECK %s
diff --git a/llvm/test/CodeGen/AMDGPU/splitkit-copy-bundle.mir b/llvm/test/CodeGen/AMDGPU/splitkit-copy-bundle.mir
index 4dfa883546040..b9a03a5965892 100644
--- a/llvm/test/CodeGen/AMDGPU/splitkit-copy-bundle.mir
+++ b/llvm/test/CodeGen/AMDGPU/splitkit-copy-bundle.mir
@@ -1,3 +1,30 @@
+# RUN: llc -mtriple=amdgpu9.00 -run-pass=greedy -enable-partial-spills -verify-machineinstrs -verify-regalloc -o - %s | FileCheck %s --check-prefix=PARTIAL
+# RUN: llc -mtriple=amdgpu9.00 -run-pass=greedy,virtregrewriter,post-RA-sched -enable-partial-spills -verify-machineinstrs -verify-regalloc -o /dev/null %s
+
+# Only the two seed words are needed across the loop. The sparse copy bundle
+# below must keep each of its eight live words at the original slot offset.
+# PARTIAL-LABEL: name: splitkit_copy_bundle
+# PARTIAL: bb.0:
+# PARTIAL: SI_SPILL_S32_SAVE_partial [[SEED:%[0-9]+]].sub0, [[PAIR:%stack.[0-9]+]], 0,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[SEED]].sub1, [[PAIR]], 4,
+# PARTIAL: SI_SPILL_S32_SAVE_partial {{%[0-9]+}}.sub0, [[SINGLE:%stack.[0-9]+]], 0,
+# PARTIAL: bb.1:
+# PARTIAL: SI_SPILL_S1024_RESTORE [[PAIR]],
+# PARTIAL: SI_SPILL_S32_SAVE_partial [[SEED:%[0-9]+]].sub0, [[PAIR]], 0,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[SEED]].sub1, [[PAIR]], 4,
+# PARTIAL: SI_SPILL_S1024_RESTORE [[SINGLE]],
+# PARTIAL: SI_SPILL_S32_SAVE_partial {{%[0-9]+}}.sub0, [[SINGLE]], 0,
+# PARTIAL-LABEL: name: splitkit_copy_unbundle_reorder
+# PARTIAL: SI_SPILL_S32_SAVE_partial [[SPARSE:%[0-9]+]].sub4, [[SLOT:%stack.[0-9]+]], 16,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[SPARSE]].sub5, [[SLOT]], 20,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[SPARSE]].sub7, [[SLOT]], 28,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[SPARSE]].sub8, [[SLOT]], 32,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[SPARSE]].sub10, [[SLOT]], 40,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[SPARSE]].sub11, [[SLOT]], 44,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[SPARSE]].sub13, [[SLOT]], 52,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial [[SPARSE]].sub14, [[SLOT]], 56,
+# PARTIAL: SI_SPILL_S512_RESTORE [[SLOT]],
+
# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py
# RUN: llc -mtriple=amdgpu9.00 -run-pass=greedy -o - -verify-machineinstrs %s | FileCheck -check-prefix=RA %s
# RUN: llc -mtriple=amdgpu9.00 -run-pass=greedy,virtregrewriter,post-RA-sched -o - -verify-machineinstrs %s | FileCheck -check-prefix=VR %s
diff --git a/llvm/test/CodeGen/AMDGPU/splitkit-nolivesubranges.mir b/llvm/test/CodeGen/AMDGPU/splitkit-nolivesubranges.mir
index a8cf32375f88e..b98e9385d0750 100644
--- a/llvm/test/CodeGen/AMDGPU/splitkit-nolivesubranges.mir
+++ b/llvm/test/CodeGen/AMDGPU/splitkit-nolivesubranges.mir
@@ -1,3 +1,18 @@
+# RUN: llc -enable-partial-spills -verify-machineinstrs -verify-regalloc -mtriple=amdgpu6.00 -run-pass=greedy,virtregrewriter %s -o - | FileCheck %s --check-prefix=PARTIAL
+# Keep this as a splitting/progress check: IMPLICIT_DEF supplies no numerical
+# value oracle. Only the high word has a consumer after the second clobber.
+# PARTIAL-LABEL: name: func0
+# PARTIAL: $sgpr104 = S_AND_B32
+# PARTIAL-NEXT: KILL implicit-def $vcc,
+# PARTIAL-NEXT: renamable $sgpr0_sgpr1 = IMPLICIT_DEF
+# PARTIAL-NEXT: renamable $sgpr0 = IMPLICIT_DEF
+# PARTIAL-NEXT: renamable $sgpr1 = IMPLICIT_DEF
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial killed renamable $sgpr1, [[SLOT:%stack.[0-9]+]], 4, implicit $exec, implicit $sp_reg :: (store (s32) into [[SLOT]] + 4, addrspace 5)
+# PARTIAL-NEXT: KILL implicit-def $vcc,
+# PARTIAL-NEXT: renamable $sgpr0_sgpr1 = SI_SPILL_S64_RESTORE [[SLOT]],
+# PARTIAL-NEXT: $sgpr105 = S_AND_B32 killed renamable $sgpr1, renamable $sgpr1, implicit-def $scc
+# PARTIAL-NEXT: S_NOP 0, implicit $sgpr104, implicit $sgpr105
+
# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py
# RUN: llc -mtriple=amdgpu6.00 -run-pass=greedy,virtregrewriter -verify-regalloc %s -o - | FileCheck %s
diff --git a/llvm/test/CodeGen/AMDGPU/splitkit.mir b/llvm/test/CodeGen/AMDGPU/splitkit.mir
index be5fdada9c41d..029a30efd20bc 100644
--- a/llvm/test/CodeGen/AMDGPU/splitkit.mir
+++ b/llvm/test/CodeGen/AMDGPU/splitkit.mir
@@ -1,3 +1,45 @@
+# RUN: llc -o - %s -mtriple=amdgpu8.03-- -enable-partial-spills -verify-machineinstrs -verify-regalloc -run-pass=regallocbasic,virtregrewriter | FileCheck %s --check-prefix=BASIC
+
+# Basic allocation spills the first definition directly, then reloads it before
+# adding word 3. No input split relationships are required on this path either.
+# BASIC-LABEL: name: splitHoist
+# BASIC: bb.0:
+# BASIC: S_NOP 0, implicit-def renamable $sgpr0
+# BASIC-NEXT: SI_SPILL_S32_SAVE_partial killed renamable $sgpr0, [[SLOT:%stack.[0-9]+]], 0,
+# BASIC-NEXT: renamable $sgpr0_sgpr1_sgpr2_sgpr3 = SI_SPILL_S128_RESTORE [[SLOT]],
+# BASIC-NEXT: S_NOP 0, implicit-def renamable $sgpr3
+# BASIC-NEXT: SI_SPILL_S128_SAVE killed renamable $sgpr0_sgpr1_sgpr2_sgpr3, [[SLOT]],
+# BASIC: bb.3:
+# BASIC: SI_SPILL_S128_RESTORE [[SLOT]],
+# BASIC-NEXT: S_NOP 0, implicit killed renamable $sgpr0
+# BASIC: SI_SPILL_S128_RESTORE [[SLOT]],
+# BASIC-NEXT: S_NOP 0, implicit killed renamable $sgpr3
+
+# RUN: llc -o - %s -mtriple=amdgpu8.03-- -enable-partial-spills -verify-machineinstrs -verify-regalloc -run-pass=greedy,virtregrewriter | FileCheck %s --check-prefix=PARTIAL
+
+# The allocator creates the split copies. Save the two defined words at their
+# original offsets; do not hoist a full store from the sparse original value.
+# PARTIAL-LABEL: name: func0
+# PARTIAL: SI_SPILL_S32_SAVE_partial renamable $sgpr0, [[SLOT:%stack.[0-9]+]], 0,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial killed renamable $sgpr3, [[SLOT]], 12,
+# PARTIAL: SI_SPILL_S128_RESTORE [[SLOT]],
+# PARTIAL: S_NOP 0, implicit renamable $sgpr0
+# PARTIAL: S_NOP 0, implicit renamable $sgpr3
+# PARTIAL-LABEL: name: splitHoist
+# PARTIAL: bb.0:
+# PARTIAL-NOT: SI_SPILL
+# PARTIAL: bb.1:
+# PARTIAL: SI_SPILL_S32_SAVE_partial renamable $sgpr0, [[SLOT:%stack.[0-9]+]], 0,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial killed renamable $sgpr3, [[SLOT]], 12,
+# PARTIAL: SI_SPILL_S128_RESTORE [[SLOT]],
+# PARTIAL: bb.2:
+# PARTIAL: SI_SPILL_S32_SAVE_partial renamable $sgpr0, [[SLOT]], 0,
+# PARTIAL-NEXT: SI_SPILL_S32_SAVE_partial killed renamable $sgpr3, [[SLOT]], 12,
+# PARTIAL: SI_SPILL_S128_RESTORE [[SLOT]],
+# PARTIAL: bb.3:
+# PARTIAL: S_NOP 0, implicit renamable $sgpr0
+# PARTIAL: S_NOP 0, implicit renamable $sgpr3
+
# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 2
# RUN: llc -o - %s -mtriple=amdgpu8.03-- -verify-machineinstrs -run-pass=greedy,virtregrewriter | FileCheck %s
--- |
diff --git a/llvm/test/CodeGen/AMDGPU/tuple-allocation-failure.ll b/llvm/test/CodeGen/AMDGPU/tuple-allocation-failure.ll
index 8e493437b1b19..ea5576d093263 100644
--- a/llvm/test/CodeGen/AMDGPU/tuple-allocation-failure.ll
+++ b/llvm/test/CodeGen/AMDGPU/tuple-allocation-failure.ll
@@ -1,3 +1,5 @@
+; RUN: llc -enable-partial-spills -verify-machineinstrs -verify-regalloc -mtriple=amdgpu9.0a-amd-amdhsa -greedy-regclass-priority-trumps-globalness=1 -o - %s | FileCheck -check-prefixes=GFX90A,GLOBALNESS1 %s
+; RUN: llc -enable-partial-spills -verify-machineinstrs -verify-regalloc -mtriple=amdgpu9.0a-amd-amdhsa -greedy-regclass-priority-trumps-globalness=0 -o - %s | FileCheck -check-prefixes=GFX90A,GLOBALNESS0 %s
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 3
; RUN: llc -mtriple=amdgpu9.0a-amd-amdhsa -greedy-regclass-priority-trumps-globalness=1 -o - %s | FileCheck -check-prefixes=GFX90A,GLOBALNESS1 %s
; RUN: llc -mtriple=amdgpu9.0a-amd-amdhsa -greedy-regclass-priority-trumps-globalness=0 -o - %s | FileCheck -check-prefixes=GFX90A,GLOBALNESS0 %s
diff --git a/llvm/test/CodeGen/X86/fp16-spill.ll b/llvm/test/CodeGen/X86/fp16-spill.ll
index 6161009b6f563..bfaabe59cbfab 100644
--- a/llvm/test/CodeGen/X86/fp16-spill.ll
+++ b/llvm/test/CodeGen/X86/fp16-spill.ll
@@ -1,3 +1,4 @@
+; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+avx512f -enable-partial-spills -verify-machineinstrs -verify-regalloc | FileCheck %s --check-prefixes=AVX512
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py UTC_ARGS: --version 5
; RUN: llc < %s -mtriple=x86_64-unknown-unknown -verify-machineinstrs | FileCheck %s --check-prefixes=SSE2
; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+avx -verify-machineinstrs | FileCheck %s --check-prefixes=AVX
diff --git a/llvm/test/CodeGen/X86/hoist-spill-lpad.ll b/llvm/test/CodeGen/X86/hoist-spill-lpad.ll
index 19d1ca396d154..f1115e84964b7 100644
--- a/llvm/test/CodeGen/X86/hoist-spill-lpad.ll
+++ b/llvm/test/CodeGen/X86/hoist-spill-lpad.ll
@@ -1,3 +1,4 @@
+; RUN: llc -enable-partial-spills -verify-machineinstrs -verify-regalloc < %s | FileCheck %s
; RUN: llc < %s | FileCheck %s
;
; PR27612. The following spill is hoisted from two locations: the fall
More information about the llvm-branch-commits
mailing list