[llvm] [CodeGen] Eliminate redundant spills at control-flow joins (PR #226852)
Yaxun Liu via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 27 17:14:11 PDT 2026
https://github.com/yxsamliu created https://github.com/llvm/llvm-project/pull/226852
A join can spill a value that an incoming path just reloaded from the
same slot. For a reload used only by that spill:
left: v = reload slot -> left:
right: v = ... -> right: v = ...; spill v, slot
join: spill v, slot -> join:
Move the spill to the incoming path that still needs it and remove the
dead reload. If both paths already have the value in the slot, remove
the join spill without inserting another store.
>From 9314e5734bc6b4c565745ec7151d8f850aced7f8 Mon Sep 17 00:00:00 2001
From: "Yaxun (Sam) Liu" <yaxun.liu at amd.com>
Date: Sun, 27 Sep 2026 14:25:25 -0400
Subject: [PATCH] [CodeGen] Eliminate redundant spills at control-flow joins
A join can spill a value that an incoming path just reloaded from the
same slot. For a reload used only by that spill:
left: v = reload slot -> left:
right: v = ... -> right: v = ...; spill v, slot
join: spill v, slot -> join:
Move the spill to the incoming path that still needs it and remove the
dead reload. If both paths already have the value in the slot, remove
the join spill without inserting another store.
---
.../include/llvm/CodeGen/LiveDebugVariables.h | 9 +-
llvm/include/llvm/CodeGen/Spiller.h | 2 +
llvm/lib/CodeGen/InlineSpiller.cpp | 279 +++++++-
llvm/lib/CodeGen/LiveDebugVariables.cpp | 89 ++-
llvm/lib/CodeGen/RegAllocBasic.cpp | 5 +-
llvm/lib/CodeGen/RegAllocGreedy.cpp | 4 +-
.../inline-spiller-join-cleanup-debug.mir | 196 ++++++
.../AMDGPU/inline-spiller-join-cleanup.mir | 643 ++++++++++++++++++
llvm/unittests/CodeGen/CMakeLists.txt | 1 +
.../CodeGen/LiveDebugVariablesTest.cpp | 124 ++++
10 files changed, 1340 insertions(+), 12 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/inline-spiller-join-cleanup-debug.mir
create mode 100644 llvm/test/CodeGen/AMDGPU/inline-spiller-join-cleanup.mir
create mode 100644 llvm/unittests/CodeGen/LiveDebugVariablesTest.cpp
diff --git a/llvm/include/llvm/CodeGen/LiveDebugVariables.h b/llvm/include/llvm/CodeGen/LiveDebugVariables.h
index d9d2a2f6a08e6..232baf1c6170f 100644
--- a/llvm/include/llvm/CodeGen/LiveDebugVariables.h
+++ b/llvm/include/llvm/CodeGen/LiveDebugVariables.h
@@ -42,11 +42,16 @@ class LiveDebugVariables {
LLVM_ABI void analyze(MachineFunction &MF, LiveIntervals *LIS);
/// splitRegister - Move any user variables in OldReg to the live ranges in
- /// NewRegs where they are live. Mark the values as unavailable where no new
- /// register is live.
+ /// NewRegs where they are live. Keep other locations in OldReg for spill-slot
+ /// rewriting. NewRegs must not include OldReg.
LLVM_ABI void splitRegister(Register OldReg, ArrayRef<Register> NewRegs,
LiveIntervals &LIS);
+ /// Mark locations outside Reg's remaining live interval as unavailable after
+ /// deleting definitions. Call splitRegister first for any split components.
+ /// Reg must not describe a value kept in a spill slot.
+ LLVM_ABI void shrinkRegister(Register Reg);
+
/// emitDebugValues - Emit new DBG_VALUE instructions reflecting the changes
/// that happened during register allocation.
/// @param VRM Rename virtual registers according to map.
diff --git a/llvm/include/llvm/CodeGen/Spiller.h b/llvm/include/llvm/CodeGen/Spiller.h
index 5434801727454..207b9f935ceff 100644
--- a/llvm/include/llvm/CodeGen/Spiller.h
+++ b/llvm/include/llvm/CodeGen/Spiller.h
@@ -20,6 +20,7 @@ class MachineFunctionPass;
class VirtRegMap;
class VirtRegAuxInfo;
class LiveIntervals;
+class LiveDebugVariables;
class LiveRegMatrix;
class LiveStacks;
class MachineDominatorTree;
@@ -53,6 +54,7 @@ class LLVM_ABI Spiller {
LiveStacks &LSS;
MachineDominatorTree &MDT;
const MachineBlockFrequencyInfo &MBFI;
+ LiveDebugVariables *DebugVars = nullptr;
};
};
diff --git a/llvm/lib/CodeGen/InlineSpiller.cpp b/llvm/lib/CodeGen/InlineSpiller.cpp
index f3682a2e24808..619a49546f33e 100644
--- a/llvm/lib/CodeGen/InlineSpiller.cpp
+++ b/llvm/lib/CodeGen/InlineSpiller.cpp
@@ -21,6 +21,7 @@
#include "llvm/ADT/SmallPtrSet.h"
#include "llvm/ADT/SmallVector.h"
#include "llvm/ADT/Statistic.h"
+#include "llvm/CodeGen/LiveDebugVariables.h"
#include "llvm/CodeGen/LiveInterval.h"
#include "llvm/CodeGen/LiveIntervals.h"
#include "llvm/CodeGen/LiveRangeEdit.h"
@@ -29,12 +30,15 @@
#include "llvm/CodeGen/MachineBasicBlock.h"
#include "llvm/CodeGen/MachineBlockFrequencyInfo.h"
#include "llvm/CodeGen/MachineDominators.h"
+#include "llvm/CodeGen/MachineFrameInfo.h"
#include "llvm/CodeGen/MachineFunction.h"
#include "llvm/CodeGen/MachineInstr.h"
#include "llvm/CodeGen/MachineInstrBuilder.h"
#include "llvm/CodeGen/MachineInstrBundle.h"
+#include "llvm/CodeGen/MachineMemOperand.h"
#include "llvm/CodeGen/MachineOperand.h"
#include "llvm/CodeGen/MachineRegisterInfo.h"
+#include "llvm/CodeGen/PseudoSourceValue.h"
#include "llvm/CodeGen/SlotIndexes.h"
#include "llvm/CodeGen/Spiller.h"
#include "llvm/CodeGen/StackMaps.h"
@@ -75,7 +79,21 @@ RestrictStatepointRemat("restrict-statepoint-remat",
cl::init(false), cl::Hidden,
cl::desc("Restrict remat for statepoint operands"));
+static cl::opt<bool> EnableSpillJoinCleanup(
+ "enable-spill-join-cleanup", cl::Hidden, cl::init(true),
+ cl::desc("Eliminate partially redundant stores in the inline spiller"));
+
+STATISTIC(NumJoinSpillsRemoved, "Number of spill stores removed from joins");
+STATISTIC(NumJoinReloadsRemoved, "Number of dead join spill reloads removed");
+
namespace {
+struct SpillAccess {
+ Register Reg;
+ int FI;
+ TypeSize Bytes = TypeSize::getZero();
+ unsigned DataOp;
+};
+
class HoistSpillHelper : private LiveRangeEdit::Delegate {
MachineFunction &MF;
LiveIntervals &LIS;
@@ -87,6 +105,8 @@ class HoistSpillHelper : private LiveRangeEdit::Delegate {
const TargetRegisterInfo &TRI;
const MachineBlockFrequencyInfo &MBFI;
LiveRegMatrix *Matrix;
+ LiveDebugVariables *DebugVars;
+ SmallSetVector<Register, 8> ShrunkRegs;
InsertPointAnalysis IPA;
@@ -128,13 +148,25 @@ class HoistSpillHelper : private LiveRangeEdit::Delegate {
SmallVectorImpl<MachineInstr *> &SpillsToRm,
DenseMap<MachineBasicBlock *, Register> &SpillsToIns);
+ bool getSpillAccess(const MachineInstr &MI, bool IsLoad,
+ SpillAccess &Access) const;
+ bool clobbersSpillInputs(const MachineInstr &MI,
+ const MachineInstr &Store) const;
+ MachineInstr *findSpillReload(MachineBasicBlock &Pred,
+ const MachineInstr &Store,
+ const SpillAccess &Access) const;
+ bool canSpillBeforeTerminator(MachineBasicBlock &Pred,
+ const MachineInstr &Store, Register Reg) const;
+ void eliminateJoinSpills(LiveRangeEdit &Edit);
+
public:
HoistSpillHelper(const Spiller::RequiredAnalyses &Analyses,
MachineFunction &mf, VirtRegMap &vrm, LiveRegMatrix *matrix)
: MF(mf), LIS(Analyses.LIS), LSS(Analyses.LSS), MDT(Analyses.MDT),
VRM(vrm), MRI(mf.getRegInfo()), TII(*mf.getSubtarget().getInstrInfo()),
TRI(*mf.getSubtarget().getRegisterInfo()), MBFI(Analyses.MBFI),
- Matrix(matrix), IPA(LIS, mf.getNumBlockIDs()) {}
+ Matrix(matrix), DebugVars(Analyses.DebugVars),
+ IPA(LIS, mf.getNumBlockIDs()) {}
void addToMergeableSpills(MachineInstr &Spill, int StackSlot,
Register Original);
@@ -1793,6 +1825,236 @@ void HoistSpillHelper::runHoistSpills(
}
}
+bool HoistSpillHelper::getSpillAccess(const MachineInstr &MI, bool IsLoad,
+ SpillAccess &Access) const {
+ if (MI.isBundled() || MI.getFlag(MachineInstr::FrameSetup) ||
+ MI.getFlag(MachineInstr::FrameDestroy) || MI.hasOrderedMemoryRef() ||
+ MI.memoperands().size() != 1 || MI.mayLoad() != IsLoad ||
+ MI.mayStore() == IsLoad)
+ return false;
+
+ // The target queries certify a pure stack access, even when a spill pseudo
+ // carries a conservative unmodeled-side-effects flag.
+ Access.Bytes = TypeSize::getZero();
+ Access.Reg = IsLoad ? TII.isLoadFromStackSlot(MI, Access.FI, Access.Bytes)
+ : TII.isStoreToStackSlot(MI, Access.FI, Access.Bytes);
+ if (!Access.Reg || !Access.Reg.isVirtual() || !Access.Bytes ||
+ Access.Bytes.isScalable() || Access.FI < 0 ||
+ !MF.getFrameInfo().isSpillSlotObjectIndex(Access.FI))
+ return false;
+
+ const MachineMemOperand &MMO = **MI.memoperands_begin();
+ const auto *PSV =
+ dyn_cast_or_null<FixedStackPseudoSourceValue>(MMO.getPseudoValue());
+ if (!PSV || PSV->getFrameIndex() != Access.FI || MMO.getOffset() != 0 ||
+ MMO.getSize() != LocationSize::precise(Access.Bytes))
+ return false;
+
+ const TargetRegisterClass *RC = MRI.getRegClass(Access.Reg);
+ if (TRI.getRegSizeInBits(*RC) != Access.Bytes * 8)
+ return false;
+
+ // Require a complete value and no hidden definitions. Comparing the other
+ // operands below also covers target addressing and predication operands.
+ bool FoundData = false;
+ for (unsigned I = 0, E = MI.getNumOperands(); I != E; ++I) {
+ const MachineOperand &MO = MI.getOperand(I);
+ if (MO.isRegMask())
+ return false;
+ if (!MO.isReg())
+ continue;
+ if (MO.getSubReg() || MO.isUndef() || MO.isInternalRead())
+ return false;
+ if (MO.getReg() == Access.Reg) {
+ if (FoundData || MO.isDef() != IsLoad || MO.isImplicit())
+ return false;
+ FoundData = true;
+ Access.DataOp = I;
+ } else if (MO.isDef() || (MO.getReg() && (!MO.getReg().isPhysical() ||
+ !MRI.isReserved(MO.getReg())))) {
+ return false;
+ }
+ }
+ return FoundData;
+}
+
+bool HoistSpillHelper::clobbersSpillInputs(const MachineInstr &MI,
+ const MachineInstr &Store) const {
+ for (const MachineOperand &MO : Store.operands()) {
+ if (MO.isReg() && MO.getReg() && MI.modifiesRegister(MO.getReg(), &TRI))
+ return true;
+ }
+ return false;
+}
+
+MachineInstr *
+HoistSpillHelper::findSpillReload(MachineBasicBlock &Pred,
+ const MachineInstr &Store,
+ const SpillAccess &Access) const {
+ Register Reg = Access.Reg;
+ unsigned Count = 0;
+ for (MachineInstr &MI : llvm::reverse(Pred)) {
+ if (++Count > 64 || MI.isBundled() || MI.isCall() || MI.isInlineAsm() ||
+ MI.getFlag(MachineInstr::FrameSetup) ||
+ MI.getFlag(MachineInstr::FrameDestroy) || MI.mayStore() ||
+ MI.hasOrderedMemoryRef())
+ return nullptr;
+
+ if (MI.isFullCopy() && MI.getNumOperands() == 2 &&
+ MI.getOperand(0).getReg() == Reg &&
+ MI.getOperand(1).getReg().isVirtual() && !MI.getOperand(1).isUndef() &&
+ TRI.getRegSizeInBits(*MRI.getRegClass(MI.getOperand(1).getReg())) ==
+ Access.Bytes * 8) {
+ Reg = MI.getOperand(1).getReg();
+ continue;
+ }
+
+ SpillAccess Load;
+ bool IsReload = getSpillAccess(MI, true, Load);
+ if (MI.hasUnmodeledSideEffects() && !IsReload)
+ return nullptr;
+ if (IsReload && Load.Reg == Reg && Load.FI == Access.FI &&
+ Load.Bytes == Access.Bytes &&
+ MI.getNumOperands() == Store.getNumOperands()) {
+ SmallVector<const MachineOperand *, 8> LoadOps, StoreOps;
+ for (unsigned I = 0, E = MI.getNumOperands(); I != E; ++I) {
+ if (I != Load.DataOp)
+ LoadOps.push_back(&MI.getOperand(I));
+ if (I != Access.DataOp)
+ StoreOps.push_back(&Store.getOperand(I));
+ }
+ if (llvm::all_of(llvm::zip(LoadOps, StoreOps), [](const auto &Ops) {
+ return std::get<0>(Ops)->isIdenticalTo(*std::get<1>(Ops));
+ }))
+ return &MI;
+ }
+ if (MI.modifiesRegister(Reg, &TRI) || clobbersSpillInputs(MI, Store))
+ return nullptr;
+ }
+ return nullptr;
+}
+
+bool HoistSpillHelper::canSpillBeforeTerminator(MachineBasicBlock &Pred,
+ const MachineInstr &Store,
+ Register Reg) const {
+ auto Insert = Pred.getFirstTerminator();
+ SlotIndex End = LIS.getMBBEndIdx(&Pred);
+ SlotIndex Idx = Insert == Pred.end() ? End.getPrevSlot()
+ : LIS.getInstructionIndex(*Insert);
+ const LiveInterval &LI = LIS.getInterval(Reg);
+ if (!LI.liveAt(Idx) || LI.getVNInfoAt(Idx) != LI.getVNInfoBefore(End))
+ return false;
+ if (LI.hasSubRanges()) {
+ LaneBitmask Covered;
+ for (const LiveInterval::SubRange &SR : LI.subranges()) {
+ if (!SR.liveAt(Idx) || SR.getVNInfoAt(Idx) != SR.getVNInfoBefore(End))
+ return false;
+ Covered |= SR.LaneMask;
+ }
+ if (Covered != MRI.getMaxLaneMaskForVReg(Reg))
+ return false;
+ }
+ for (const MachineInstr &MI : make_range(Insert, Pred.end()))
+ if (!MI.isBranch() || MI.isBundled() || MI.hasUnmodeledSideEffects() ||
+ MI.mayLoadOrStore() || clobbersSpillInputs(MI, Store))
+ return false;
+ return true;
+}
+
+/// Distribute a join spill onto the incoming path that still needs it. The
+/// other path already has the value in the slot, so its reload may become dead.
+void HoistSpillHelper::eliminateJoinSpills(LiveRangeEdit &Edit) {
+ // Allocators using this cleanup must maintain the collected debug locations.
+ if (!DebugVars || MF.exposesReturnsTwice())
+ return;
+
+ // The dominance-based hoisting phase is finished. Do not retain its pointers
+ // while dead-definition elimination deletes or rewrites instructions.
+ MergeableSpills.clear();
+ for (MachineBasicBlock &MBB : MF) {
+ if (MBB.pred_size() != 2 || MBB.isEHPad() || MBB.isEHScopeReturnBlock() ||
+ llvm::any_of(MBB.predecessors(), [&](const auto *Pred) {
+ return Pred == &MBB || Pred->succ_size() != 1 || Pred->isEHPad();
+ }))
+ continue;
+
+ for (unsigned Count = 0; Count != 16 && !MBB.empty(); ++Count) {
+ MachineInstr &Store = MBB.front();
+ SpillAccess Access;
+ if (!getSpillAccess(Store, false, Access) ||
+ !LSS.hasInterval(Access.FI) || !LIS.hasInterval(Access.Reg) ||
+ (!VRM.hasPhys(Access.Reg) &&
+ !PendingReassignments.contains(Access.Reg)))
+ break;
+ SlotIndex StoreIdx = LIS.getInstructionIndex(Store);
+ const LiveInterval &LI = LIS.getInterval(Access.Reg);
+ VNInfo *VNI = LI.getVNInfoAt(StoreIdx);
+ if (!VNI || !VNI->isPHIDef() || VNI->def != LIS.getMBBStartIdx(&MBB))
+ break;
+
+ SmallVector<SlotIndex, 2> Reloads;
+ MachineBasicBlock *NeedsStore = nullptr;
+ bool Legal = true;
+ for (MachineBasicBlock *Pred : MBB.predecessors()) {
+ if (!canSpillBeforeTerminator(*Pred, Store, Access.Reg)) {
+ Legal = false;
+ break;
+ }
+ if (MachineInstr *Load = findSpillReload(*Pred, Store, Access))
+ Reloads.push_back(LIS.getInstructionIndex(*Load));
+ else
+ NeedsStore = Pred;
+ }
+ if (!Legal || Reloads.empty())
+ break;
+
+ LiveInterval &StackLI = LSS.getInterval(Access.FI);
+ VNInfo *StackVNI = StackLI.getValNumInfo(0);
+ // Preserve the slot through the join after removing its store, including
+ // gaps where a predecessor's reload used to end the slot's lifetime.
+ StackLI.addSegment(LiveRange::Segment(LIS.getMBBStartIdx(&MBB),
+ StoreIdx.getRegSlot(), StackVNI));
+ for (SlotIndex Idx : Reloads)
+ StackLI.addSegment(LiveRange::Segment(
+ Idx.getRegSlot(),
+ LIS.getMBBEndIdx(LIS.getInstructionFromIndex(Idx)->getParent()),
+ StackVNI));
+ if (NeedsStore) {
+ MachineInstr *Clone = MF.CloneMachineInstr(&Store);
+ Clone->clearKillInfo();
+ NeedsStore->insert(NeedsStore->getFirstTerminator(), Clone);
+ SlotIndex NewIdx = LIS.InsertMachineInstrInMaps(*Clone);
+ // The store starts the slot's lifetime on this predecessor. Its old
+ // live range need not cover the incoming value of the register.
+ StackLI.addSegment(LiveRange::Segment(
+ NewIdx.getRegSlot(), LIS.getMBBEndIdx(NeedsStore), StackVNI));
+ }
+ LLVM_DEBUG(dbgs() << "Removing join spill: " << Store);
+ Store.setDesc(TII.get(TargetOpcode::KILL));
+ SmallVector<MachineInstr *, 2> Dead{&Store};
+ Edit.eliminateDeadDefs(Dead, {});
+ ++NumJoinSpillsRemoved;
+
+ for (SlotIndex Idx : Reloads) {
+ MachineInstr *Load = LIS.getInstructionFromIndex(Idx);
+ if (!Load) {
+ ++NumJoinReloadsRemoved;
+ continue;
+ }
+ if (!Load->allDefsAreDead())
+ continue;
+ // Target-certified pure reload pseudos can carry conservative side
+ // effect flags. KILL lets LiveRangeEdit remove their dead definitions.
+ LLVM_DEBUG(dbgs() << "Removing join reload: " << *Load);
+ Load->setDesc(TII.get(TargetOpcode::KILL));
+ Dead.push_back(Load);
+ Edit.eliminateDeadDefs(Dead, {});
+ ++NumJoinReloadsRemoved;
+ }
+ }
+ }
+}
+
/// For spills with equal values, remove redundant spills and hoist those left
/// to less hot spots.
///
@@ -1887,6 +2149,9 @@ void HoistSpillHelper::hoistAllSpills() {
Edit.eliminateDeadDefs(SpillsToRm, {});
}
+ if (EnableSpillJoinCleanup)
+ eliminateJoinSpills(Edit);
+
// Flush vregs that were unassigned from the matrix during shrinking but
// were not split (so LRE_DidCloneVirtReg never re-assigned them).
for (auto &[VReg, PhysReg] : PendingReassignments) {
@@ -1895,6 +2160,10 @@ void HoistSpillHelper::hoistAllSpills() {
Matrix->assign(LIS.getInterval(VReg), PhysReg);
}
PendingReassignments.clear();
+
+ for (Register Reg : ShrunkRegs)
+ DebugVars->shrinkRegister(Reg);
+ ShrunkRegs.clear();
}
/// Called when a virtual register's live interval is about to be shrunk.
@@ -1902,6 +2171,9 @@ void HoistSpillHelper::hoistAllSpills() {
/// later LRE_DidCloneVirtReg or by hoistAllSpills' flush, and stash the
/// physreg in PendingReassignments since the unassign clears VRM.
void HoistSpillHelper::LRE_WillShrinkVirtReg(Register VirtReg) {
+ if (EnableSpillJoinCleanup && DebugVars &&
+ (VRM.hasPhys(VirtReg) || PendingReassignments.contains(VirtReg)))
+ ShrunkRegs.insert(VirtReg);
if (!Matrix || !VRM.hasPhys(VirtReg) || !LIS.hasInterval(VirtReg))
return;
@@ -1915,6 +2187,9 @@ void HoistSpillHelper::LRE_WillShrinkVirtReg(Register VirtReg) {
/// Forcibly remove the register from LiveRegMatrix before it's deleted,
/// preventing dangling pointers.
bool HoistSpillHelper::LRE_CanEraseVirtReg(Register VirtReg) {
+ if (EnableSpillJoinCleanup && DebugVars &&
+ (VRM.hasPhys(VirtReg) || PendingReassignments.contains(VirtReg)))
+ ShrunkRegs.insert(VirtReg);
PendingReassignments.erase(VirtReg);
if (Matrix && VRM.hasPhys(VirtReg)) {
const LiveInterval &LI = LIS.getInterval(VirtReg);
@@ -1926,6 +2201,8 @@ bool HoistSpillHelper::LRE_CanEraseVirtReg(Register VirtReg) {
/// For VirtReg clone, the \p New register should have the same physreg or
/// stackslot as the \p old register.
void HoistSpillHelper::LRE_DidCloneVirtReg(Register New, Register Old) {
+ if (EnableSpillJoinCleanup && DebugVars)
+ DebugVars->splitRegister(Old, {New}, LIS);
// New is freshly created by LiveRangeEdit::eliminateDeadDefs and its interval
// is guaranteed to exist on every path below.
assert(LIS.hasInterval(New) && "Cloned vreg without live interval");
diff --git a/llvm/lib/CodeGen/LiveDebugVariables.cpp b/llvm/lib/CodeGen/LiveDebugVariables.cpp
index b4347cad2c1b2..e954f9db97cb0 100644
--- a/llvm/lib/CodeGen/LiveDebugVariables.cpp
+++ b/llvm/lib/CodeGen/LiveDebugVariables.cpp
@@ -61,6 +61,7 @@
#include <map>
#include <memory>
#include <optional>
+#include <tuple>
#include <utility>
using namespace llvm;
@@ -472,6 +473,8 @@ class UserValue {
bool splitRegister(Register OldReg, ArrayRef<Register> NewRegs,
LiveIntervals &LIS);
+ void shrinkRegister(Register Reg, const LiveInterval *LI);
+
/// Rewrite virtual register locations according to the provided virtual
/// register map. Record the stack slot offsets for the locations that
/// were spilled.
@@ -665,6 +668,8 @@ class LiveDebugVariables::LDVImpl {
/// Replace all references to OldReg with NewRegs.
void splitRegister(Register OldReg, ArrayRef<Register> NewRegs);
+ void shrinkRegister(Register Reg);
+
/// Recreate DBG_VALUE instruction from data structures.
void emitDebugValues(VirtRegMap *VRM);
@@ -1484,6 +1489,80 @@ UserValue::splitRegister(Register OldReg, ArrayRef<Register> NewRegs,
return DidChange;
}
+void UserValue::shrinkRegister(Register Reg, const LiveInterval *LI) {
+ auto IsRegLocation = [&](unsigned LocNo) {
+ return LocNo != UndefLocNo && locations[LocNo].isReg() &&
+ locations[LocNo].getReg() == Reg;
+ };
+ bool Changed = false;
+ for (LocMap::iterator It = locInts.begin(); It.valid();) {
+ SlotIndex Start = It.start(), Stop = It.stop();
+ auto Seg = LI ? LI->find(Start) : LiveRange::const_iterator();
+ if (!llvm::any_of(It.value().loc_nos(), IsRegLocation) ||
+ (LI && Seg != LI->end() && Seg->start <= Start && Seg->end >= Stop)) {
+ ++It;
+ continue;
+ }
+
+ DbgVariableValue Value = It.value();
+ SmallVector<unsigned, 4> LocNos;
+ for (unsigned LocNo : Value.loc_nos())
+ LocNos.push_back(IsRegLocation(LocNo) ? UndefLocNo : LocNo);
+ DbgVariableValue Unavailable(LocNos, Value.getWasIndirect(),
+ Value.getWasList(), *Value.getExpression());
+
+ // Replace only affected intervals. All locations for Reg share these gaps,
+ // including multiple subregister operands in a DBG_VALUE_LIST.
+ It.erase();
+ if (LI) {
+ for (; Seg != LI->end() && Seg->start < Stop; ++Seg) {
+ if (Start < Seg->start) {
+ locInts.insert(Start, Seg->start, Unavailable);
+ Start = Seg->start;
+ }
+ SlotIndex End = std::min(Stop, Seg->end);
+ locInts.insert(Start, End, Value);
+ Start = End;
+ }
+ }
+ if (Start < Stop)
+ locInts.insert(Start, Stop, Unavailable);
+ // Insertion can invalidate the iterator or coalesce with its neighbors.
+ It.find(Stop);
+ Changed = true;
+ }
+
+ if (Changed)
+ for (unsigned I = locations.size(); I; --I)
+ if (IsRegLocation(I - 1))
+ removeLocationIfUnused(I - 1);
+}
+
+void LiveDebugVariables::LDVImpl::shrinkRegister(Register Reg) {
+ const LiveInterval *LI =
+ LIS->hasInterval(Reg) ? &LIS->getInterval(Reg) : nullptr;
+ for (UserValue *UV = lookupVirtReg(Reg); UV; UV = UV->getNext())
+ UV->shrinkRegister(Reg, LI);
+
+ auto It = RegToPHIIdx.find(Reg);
+ if (It == RegToPHIIdx.end())
+ return;
+ llvm::erase_if(It->second, [&](unsigned InstrID) {
+ auto PHIIt = PHIValToPos.find(InstrID);
+ if (LI && LI->liveAt(PHIIt->second.SI))
+ return false;
+ PHIValToPos.erase(PHIIt);
+ return true;
+ });
+ if (It->second.empty())
+ RegToPHIIdx.erase(It);
+}
+
+void LiveDebugVariables::shrinkRegister(Register Reg) {
+ if (PImpl)
+ PImpl->shrinkRegister(Reg);
+}
+
void LiveDebugVariables::LDVImpl::splitPHIRegister(Register OldReg,
ArrayRef<Register> NewRegs) {
auto RegIt = RegToPHIIdx.find(OldReg);
@@ -1499,22 +1578,22 @@ void LiveDebugVariables::LDVImpl::splitPHIRegister(Register OldReg,
assert(OldReg == PHIIt->second.Reg);
// Find the new register that covers this position.
+ Register NewPHIReg = OldReg;
for (auto NewReg : NewRegs) {
const LiveInterval &LI = LIS->getInterval(NewReg);
auto LII = LI.find(Slot);
if (LII != LI.end() && LII->start <= Slot) {
// This new register covers this PHI position, record this for indexing.
- NewRegIdxes.push_back(std::make_pair(NewReg, InstrID));
+ NewPHIReg = NewReg;
// Record that this value lives in a different VReg now.
PHIIt->second.Reg = NewReg;
break;
}
}
- // If we do not find a new register covering this PHI, then register
- // allocation has dropped its location, for example because it's not live.
- // The old VReg will not be mapped to a physreg, and the instruction
- // number will have been optimized out.
+ // Keep indexing unmatched PHIs by OldReg for spill-slot rewriting or a
+ // later shrink/split of a surviving component.
+ NewRegIdxes.emplace_back(NewPHIReg, InstrID);
}
// Re-create register index using the new register numbers.
diff --git a/llvm/lib/CodeGen/RegAllocBasic.cpp b/llvm/lib/CodeGen/RegAllocBasic.cpp
index 0d8a6970b41e7..c6cb4f414e215 100644
--- a/llvm/lib/CodeGen/RegAllocBasic.cpp
+++ b/llvm/lib/CodeGen/RegAllocBasic.cpp
@@ -225,6 +225,7 @@ bool RABasic::runOnMachineFunction(MachineFunction &mf) {
auto &MBFI = getAnalysis<MachineBlockFrequencyInfoWrapperPass>().getMBFI();
auto &LiveStks = getAnalysis<LiveStacksWrapperLegacy>().getLS();
auto &MDT = getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
+ auto &DebugVars = getAnalysis<LiveDebugVariablesWrapperLegacy>().getLDV();
RegAllocBase::init(getAnalysis<VirtRegMapWrapperLegacy>().getVRM(),
getAnalysis<LiveIntervalsWrapperPass>().getLIS(),
@@ -234,8 +235,8 @@ bool RABasic::runOnMachineFunction(MachineFunction &mf) {
&getAnalysis<ProfileSummaryInfoWrapperPass>().getPSI());
VRAI.calculateSpillWeightsAndHints();
- SpillerInstance.reset(
- createInlineSpiller({*LIS, LiveStks, MDT, MBFI}, *MF, *VRM, VRAI));
+ SpillerInstance.reset(createInlineSpiller(
+ {*LIS, LiveStks, MDT, MBFI, &DebugVars}, *MF, *VRM, VRAI));
allocatePhysRegs();
postOptimization();
diff --git a/llvm/lib/CodeGen/RegAllocGreedy.cpp b/llvm/lib/CodeGen/RegAllocGreedy.cpp
index af4fc60fe5ec3..a47d839469223 100644
--- a/llvm/lib/CodeGen/RegAllocGreedy.cpp
+++ b/llvm/lib/CodeGen/RegAllocGreedy.cpp
@@ -2981,8 +2981,8 @@ bool RAGreedy::run(MachineFunction &mf) {
PriorityAdvisor = PriorityProvider->getAdvisor(*MF, *this, *Indexes);
VRAI = std::make_unique<VirtRegAuxInfo>(*MF, *LIS, *VRM, *Loops, *MBFI);
- SpillerInstance.reset(createInlineSpiller({*LIS, *LSS, *DomTree, *MBFI}, *MF,
- *VRM, *VRAI, Matrix));
+ SpillerInstance.reset(createInlineSpiller(
+ {*LIS, *LSS, *DomTree, *MBFI, DebugVars}, *MF, *VRM, *VRAI, Matrix));
VRAI->calculateSpillWeightsAndHints();
diff --git a/llvm/test/CodeGen/AMDGPU/inline-spiller-join-cleanup-debug.mir b/llvm/test/CodeGen/AMDGPU/inline-spiller-join-cleanup-debug.mir
new file mode 100644
index 0000000000000..e4aeaaa90084b
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/inline-spiller-join-cleanup-debug.mir
@@ -0,0 +1,196 @@
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -run-pass=greedy,virtregrewriter -stress-regalloc=4 -verify-machineinstrs -verify-regalloc -enable-spill-join-cleanup=false %s -o - | FileCheck %s --check-prefixes=CHECK,OFF
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -run-pass=greedy,virtregrewriter -stress-regalloc=4 -verify-machineinstrs -verify-regalloc %s -o - | FileCheck %s --check-prefixes=CHECK,ON
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -passes=greedy,virt-reg-rewriter -stress-regalloc=4 -verify-machineinstrs -verify-regalloc -enable-spill-join-cleanup=false %s -o - | FileCheck %s --check-prefixes=CHECK,OFF
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -passes=greedy,virt-reg-rewriter -stress-regalloc=4 -verify-machineinstrs -verify-regalloc %s -o - | FileCheck %s --check-prefixes=CHECK,ON
+# The allocator creates a split register with definitions on both incoming
+# paths. Cleanup deletes the loop-exit reload but keeps the other definition.
+# Check the emitted locations: neither a whole register, a subregister, nor a
+# list operand may name the removed value's physical register. Focused checks
+# keep the debug-location contract visible among the unrelated spill traffic.
+
+# CHECK-LABEL: name: partial_join
+# CHECK: bb.2:
+# CHECK: DBG_VALUE $vgpr2_vgpr3_vgpr4_vgpr5, $noreg,
+# CHECK: DBG_VALUE $vgpr3, $noreg,
+# CHECK: DBG_VALUE_LIST
+# CHECK: bb.3:
+# CHECK: DBG_VALUE [[SLOT:%stack.[0-9]+]], 0,
+# CHECK: SI_SPILL_AV128_SAVE
+# OFF: SI_SPILL_AV128_RESTORE [[SLOT]]
+# OFF: DBG_VALUE $vgpr2_vgpr3_vgpr4_vgpr5, $noreg,
+# OFF-NEXT: DBG_VALUE $vgpr3, $noreg,
+# OFF-NEXT: DBG_VALUE_LIST {{.*}}, $vgpr2, $vgpr3,
+# ON-NOT: SI_SPILL_AV128_RESTORE
+# ON-NOT: DBG_VALUE $vgpr
+# ON-NEXT: DBG_VALUE $noreg, $noreg,
+# ON-NEXT: DBG_VALUE $noreg, $noreg,
+# ON-NEXT: DBG_VALUE_LIST {{.*}}, $noreg,
+# CHECK-NEXT: S_BRANCH %bb.4
+
+--- |
+ define amdgpu_kernel void @partial_join() !dbg !4 { ret void }
+ !llvm.dbg.cu = !{!0}
+ !llvm.module.flags = !{!9}
+ !0 = distinct !DICompileUnit(language: DW_LANG_C, file: !1, producer: "llvm", isOptimized: true, runtimeVersion: 0, emissionKind: FullDebug)
+ !1 = !DIFile(filename: "test.c", directory: "/")
+ !2 = !DISubroutineType(types: !3)
+ !3 = !{}
+ !4 = distinct !DISubprogram(name: "partial_join", scope: !1, file: !1, line: 1, type: !2, scopeLine: 1, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0, retainedNodes: !3)
+ !5 = !DIBasicType(name: "accumulator", size: 128, encoding: DW_ATE_signed)
+ !6 = !DILocalVariable(name: "acc", scope: !4, file: !1, line: 2, type: !5)
+ !10 = !DIBasicType(name: "int", size: 32, encoding: DW_ATE_signed)
+ !11 = !DILocalVariable(name: "component", scope: !4, file: !1, line: 2, type: !10)
+ !12 = !DILocalVariable(name: "sum", scope: !4, file: !1, line: 2, type: !10)
+ !8 = !DILocation(line: 2, column: 1, scope: !4)
+ !9 = !{i32 2, !"Debug Info Version", i32 3}
+...
+---
+name: partial_join
+tracksRegLiveness: true
+machineFunctionInfo:
+ stackPtrOffsetReg: '$sgpr32'
+body: |
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $vgpr0_vgpr1, $vgpr2, $sgpr7, $vgpr10_vgpr11_vgpr12_vgpr13
+ $sgpr6 = S_MOV_B32 4
+ %0:vreg_64_align2 = COPY $vgpr0_vgpr1
+ %1:vgpr_32 = COPY $vgpr2
+ %2:vgpr_32 = V_ADD_U32_e32 1, %1, implicit $exec
+ %3:vgpr_32 = V_ADD_U32_e32 2, %1, implicit $exec
+ %4:vgpr_32 = V_ADD_U32_e32 3, %1, implicit $exec
+ %5:vgpr_32 = V_ADD_U32_e32 4, %1, implicit $exec
+ undef %6.sub0:vreg_128_align2 = COPY %2
+ %6.sub1:vreg_128_align2 = COPY %3
+ %6.sub2:vreg_128_align2 = COPY %4
+ %6.sub3:vreg_128_align2 = COPY %5
+ %7:vgpr_32 = V_ADD_U32_e32 5, %1, implicit $exec
+ %8:vgpr_32 = V_ADD_U32_e32 6, %1, implicit $exec
+ %9:vgpr_32 = V_ADD_U32_e32 7, %1, implicit $exec
+ %10:vgpr_32 = V_ADD_U32_e32 8, %1, implicit $exec
+ undef %11.sub0:vreg_128_align2 = COPY %7
+ %11.sub1:vreg_128_align2 = COPY %8
+ %11.sub2:vreg_128_align2 = COPY %9
+ %11.sub3:vreg_128_align2 = COPY %10
+ %12:vgpr_32 = V_ADD_U32_e32 9, %1, implicit $exec
+ %13:vgpr_32 = V_ADD_U32_e32 10, %1, implicit $exec
+ %14:vgpr_32 = V_ADD_U32_e32 11, %1, implicit $exec
+ %15:vgpr_32 = V_ADD_U32_e32 12, %1, implicit $exec
+ undef %16.sub0:vreg_128_align2 = COPY %12
+ %16.sub1:vreg_128_align2 = COPY %13
+ %16.sub2:vreg_128_align2 = COPY %14
+ %16.sub3:vreg_128_align2 = COPY %15
+ %17:vgpr_32 = V_ADD_U32_e32 13, %1, implicit $exec
+ %18:vgpr_32 = V_ADD_U32_e32 14, %1, implicit $exec
+ %19:vgpr_32 = V_ADD_U32_e32 15, %1, implicit $exec
+ %20:vgpr_32 = V_ADD_U32_e32 16, %1, implicit $exec
+ undef %21.sub0:vreg_128_align2 = COPY %17
+ %21.sub1:vreg_128_align2 = COPY %18
+ %21.sub2:vreg_128_align2 = COPY %19
+ %21.sub3:vreg_128_align2 = COPY %20
+ S_CMP_EQ_U32 $sgpr7, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.2, implicit $scc
+ S_BRANCH %bb.1
+ bb.1:
+ successors: %bb.1(0x78000000), %bb.3(0x08000000)
+ liveins: $sgpr7, $sgpr6
+ %6.sub0:vreg_128_align2 = V_ADD_U32_e32 1, %6.sub0, implicit $exec
+ %6.sub1:vreg_128_align2 = V_ADD_U32_e32 2, %6.sub1, implicit $exec
+ %6.sub2:vreg_128_align2 = V_ADD_U32_e32 3, %6.sub2, implicit $exec
+ %6.sub3:vreg_128_align2 = V_ADD_U32_e32 4, %6.sub3, implicit $exec
+ %11.sub0:vreg_128_align2 = V_ADD_U32_e32 2, %11.sub0, implicit $exec
+ %11.sub1:vreg_128_align2 = V_ADD_U32_e32 3, %11.sub1, implicit $exec
+ %11.sub2:vreg_128_align2 = V_ADD_U32_e32 4, %11.sub2, implicit $exec
+ %11.sub3:vreg_128_align2 = V_ADD_U32_e32 5, %11.sub3, implicit $exec
+ %16.sub0:vreg_128_align2 = V_ADD_U32_e32 3, %16.sub0, implicit $exec
+ %16.sub1:vreg_128_align2 = V_ADD_U32_e32 4, %16.sub1, implicit $exec
+ %16.sub2:vreg_128_align2 = V_ADD_U32_e32 5, %16.sub2, implicit $exec
+ %16.sub3:vreg_128_align2 = V_ADD_U32_e32 6, %16.sub3, implicit $exec
+ %21.sub0:vreg_128_align2 = V_ADD_U32_e32 4, %21.sub0, implicit $exec
+ %21.sub1:vreg_128_align2 = V_ADD_U32_e32 5, %21.sub1, implicit $exec
+ %21.sub2:vreg_128_align2 = V_ADD_U32_e32 6, %21.sub2, implicit $exec
+ %21.sub3:vreg_128_align2 = V_ADD_U32_e32 7, %21.sub3, implicit $exec
+ $sgpr7 = S_SUB_U32 $sgpr7, 1, implicit-def $scc
+ S_CMP_LG_U32 $sgpr7, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.1, implicit $scc
+ S_BRANCH %bb.3
+ bb.2:
+ successors: %bb.4
+ liveins: $sgpr6, $vgpr10_vgpr11_vgpr12_vgpr13
+ %11:vreg_128_align2 = COPY $vgpr10_vgpr11_vgpr12_vgpr13
+ DBG_VALUE %6, $noreg, !6, !DIExpression(), debug-location !8
+ DBG_VALUE %6.sub1, $noreg, !11, !DIExpression(), debug-location !8
+ DBG_VALUE_LIST !12, !DIExpression(DW_OP_LLVM_arg, 0, DW_OP_LLVM_arg, 1, DW_OP_plus, DW_OP_stack_value), %6.sub0, %6.sub1, debug-location !8
+ S_BRANCH %bb.4, debug-location !8
+ bb.3:
+ successors: %bb.4
+ liveins: $sgpr6
+ DBG_VALUE %6, $noreg, !6, !DIExpression(), debug-location !8
+ DBG_VALUE %6.sub1, $noreg, !11, !DIExpression(), debug-location !8
+ DBG_VALUE_LIST !12, !DIExpression(DW_OP_LLVM_arg, 0, DW_OP_LLVM_arg, 1, DW_OP_plus, DW_OP_stack_value), %6.sub0, %6.sub1, debug-location !8
+ S_BRANCH %bb.4, debug-location !8
+ bb.4:
+ liveins: $sgpr6
+ successors: %bb.5
+ DBG_VALUE %6, $noreg, !6, !DIExpression(), debug-location !8
+ DBG_VALUE %6.sub1, $noreg, !11, !DIExpression(), debug-location !8
+ DBG_VALUE_LIST !12, !DIExpression(DW_OP_LLVM_arg, 0, DW_OP_LLVM_arg, 1, DW_OP_plus, DW_OP_stack_value), %6.sub0, %6.sub1, debug-location !8
+ S_BRANCH %bb.5, debug-location !8
+ bb.5:
+ successors: %bb.5(0x78000000), %bb.6(0x08000000)
+ liveins: $sgpr6
+ %22:vgpr_32 = GLOBAL_LOAD_DWORD %0, 0, 0, implicit $exec :: (load (s32), addrspace 1)
+ %23:vgpr_32 = V_ADD_U32_e32 %22, %6.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %23, 0, 0, implicit $exec :: (store (s32), addrspace 1)
+ %24:vgpr_32 = GLOBAL_LOAD_DWORD %0, 4, 0, implicit $exec :: (load (s32), addrspace 1)
+ %25:vgpr_32 = V_ADD_U32_e32 %24, %6.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %25, 4, 0, implicit $exec :: (store (s32), addrspace 1)
+ %26:vgpr_32 = GLOBAL_LOAD_DWORD %0, 8, 0, implicit $exec :: (load (s32), addrspace 1)
+ %27:vgpr_32 = V_ADD_U32_e32 %26, %6.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %27, 8, 0, implicit $exec :: (store (s32), addrspace 1)
+ %28:vgpr_32 = GLOBAL_LOAD_DWORD %0, 12, 0, implicit $exec :: (load (s32), addrspace 1)
+ %29:vgpr_32 = V_ADD_U32_e32 %28, %6.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %29, 12, 0, implicit $exec :: (store (s32), addrspace 1)
+ %30:vgpr_32 = GLOBAL_LOAD_DWORD %0, 16, 0, implicit $exec :: (load (s32), addrspace 1)
+ %31:vgpr_32 = V_ADD_U32_e32 %30, %11.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %31, 16, 0, implicit $exec :: (store (s32), addrspace 1)
+ %32:vgpr_32 = GLOBAL_LOAD_DWORD %0, 20, 0, implicit $exec :: (load (s32), addrspace 1)
+ %33:vgpr_32 = V_ADD_U32_e32 %32, %11.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %33, 20, 0, implicit $exec :: (store (s32), addrspace 1)
+ %34:vgpr_32 = GLOBAL_LOAD_DWORD %0, 24, 0, implicit $exec :: (load (s32), addrspace 1)
+ %35:vgpr_32 = V_ADD_U32_e32 %34, %11.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %35, 24, 0, implicit $exec :: (store (s32), addrspace 1)
+ %36:vgpr_32 = GLOBAL_LOAD_DWORD %0, 28, 0, implicit $exec :: (load (s32), addrspace 1)
+ %37:vgpr_32 = V_ADD_U32_e32 %36, %11.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %37, 28, 0, implicit $exec :: (store (s32), addrspace 1)
+ %38:vgpr_32 = GLOBAL_LOAD_DWORD %0, 32, 0, implicit $exec :: (load (s32), addrspace 1)
+ %39:vgpr_32 = V_ADD_U32_e32 %38, %16.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %39, 32, 0, implicit $exec :: (store (s32), addrspace 1)
+ %40:vgpr_32 = GLOBAL_LOAD_DWORD %0, 36, 0, implicit $exec :: (load (s32), addrspace 1)
+ %41:vgpr_32 = V_ADD_U32_e32 %40, %16.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %41, 36, 0, implicit $exec :: (store (s32), addrspace 1)
+ %42:vgpr_32 = GLOBAL_LOAD_DWORD %0, 40, 0, implicit $exec :: (load (s32), addrspace 1)
+ %43:vgpr_32 = V_ADD_U32_e32 %42, %16.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %43, 40, 0, implicit $exec :: (store (s32), addrspace 1)
+ %44:vgpr_32 = GLOBAL_LOAD_DWORD %0, 44, 0, implicit $exec :: (load (s32), addrspace 1)
+ %45:vgpr_32 = V_ADD_U32_e32 %44, %16.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %45, 44, 0, implicit $exec :: (store (s32), addrspace 1)
+ %46:vgpr_32 = GLOBAL_LOAD_DWORD %0, 48, 0, implicit $exec :: (load (s32), addrspace 1)
+ %47:vgpr_32 = V_ADD_U32_e32 %46, %21.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %47, 48, 0, implicit $exec :: (store (s32), addrspace 1)
+ %48:vgpr_32 = GLOBAL_LOAD_DWORD %0, 52, 0, implicit $exec :: (load (s32), addrspace 1)
+ %49:vgpr_32 = V_ADD_U32_e32 %48, %21.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %49, 52, 0, implicit $exec :: (store (s32), addrspace 1)
+ %50:vgpr_32 = GLOBAL_LOAD_DWORD %0, 56, 0, implicit $exec :: (load (s32), addrspace 1)
+ %51:vgpr_32 = V_ADD_U32_e32 %50, %21.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %51, 56, 0, implicit $exec :: (store (s32), addrspace 1)
+ %52:vgpr_32 = GLOBAL_LOAD_DWORD %0, 60, 0, implicit $exec :: (load (s32), addrspace 1)
+ %53:vgpr_32 = V_ADD_U32_e32 %52, %21.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %53, 60, 0, implicit $exec :: (store (s32), addrspace 1)
+ $sgpr6 = S_SUB_U32 $sgpr6, 1, implicit-def $scc
+ S_CMP_LG_U32 $sgpr6, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.5, implicit $scc
+ S_BRANCH %bb.6
+ bb.6:
+ S_ENDPGM 0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/inline-spiller-join-cleanup.mir b/llvm/test/CodeGen/AMDGPU/inline-spiller-join-cleanup.mir
new file mode 100644
index 0000000000000..78ffae2f52dae
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/inline-spiller-join-cleanup.mir
@@ -0,0 +1,643 @@
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -run-pass=greedy -stress-regalloc=4 -verify-machineinstrs -verify-regalloc -enable-spill-join-cleanup=false %s -o - | FileCheck %s --check-prefixes=CHECK,OFF
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -run-pass=greedy -stress-regalloc=4 -verify-machineinstrs -verify-regalloc %s -o - | FileCheck %s --check-prefixes=CHECK,ON
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -passes=greedy -stress-regalloc=4 -verify-machineinstrs -verify-regalloc %s -o - | FileCheck %s --check-prefixes=CHECK,ON
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -passes=greedy,virt-reg-rewriter,stack-slot-coloring -stress-regalloc=4 -verify-machineinstrs -verify-regalloc %s -o %t
+
+# Constructed from an optional loop followed by a second loop that consumes its
+# accumulators. All values are initialized; greedy creates the spill slots and
+# split ancestry. Each tuple is fully initialized by its four subregister
+# definitions. Check only the join and its predecessors.
+
+# Both incoming values come from the same slot. Remove the two reloads and
+# the join store, retaining the unrelated address reload.
+# CHECK-LABEL: name: full_join
+# CHECK: bb.2:
+# CHECK: SI_SPILL_AV128_SAVE
+# OFF: [[F:%[0-9]+]]:av_128_align2 = SI_SPILL_AV128_RESTORE [[FS:%stack.[0-9]+]],
+# ON-NOT: SI_SPILL
+# CHECK: S_BRANCH %bb.4
+# CHECK: bb.3:
+# CHECK: SI_SPILL_AV128_SAVE
+# OFF: [[F]]:av_128_align2 = SI_SPILL_AV128_RESTORE [[FS]],
+# ON-NOT: SI_SPILL
+# CHECK: S_BRANCH %bb.4
+# CHECK: bb.4:
+# OFF: SI_SPILL_AV128_SAVE [[F]], [[FS]],
+# ON-NOT: SI_SPILL_AV128_SAVE
+# CHECK: SI_SPILL_AV64_RESTORE
+# CHECK-NEXT: S_BRANCH %bb.5
+
+---
+name: full_join
+tracksRegLiveness: true
+machineFunctionInfo:
+ stackPtrOffsetReg: '$sgpr32'
+body: |
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $vgpr0_vgpr1, $vgpr2, $sgpr7
+ $sgpr6 = S_MOV_B32 4
+ %0:vreg_64_align2 = COPY $vgpr0_vgpr1
+ %1:vgpr_32 = COPY $vgpr2
+ %2:vgpr_32 = V_ADD_U32_e32 1, %1, implicit $exec
+ %3:vgpr_32 = V_ADD_U32_e32 2, %1, implicit $exec
+ %4:vgpr_32 = V_ADD_U32_e32 3, %1, implicit $exec
+ %5:vgpr_32 = V_ADD_U32_e32 4, %1, implicit $exec
+ undef %6.sub0:vreg_128_align2 = COPY %2
+ %6.sub1:vreg_128_align2 = COPY %3
+ %6.sub2:vreg_128_align2 = COPY %4
+ %6.sub3:vreg_128_align2 = COPY %5
+ %7:vgpr_32 = V_ADD_U32_e32 5, %1, implicit $exec
+ %8:vgpr_32 = V_ADD_U32_e32 6, %1, implicit $exec
+ %9:vgpr_32 = V_ADD_U32_e32 7, %1, implicit $exec
+ %10:vgpr_32 = V_ADD_U32_e32 8, %1, implicit $exec
+ undef %11.sub0:vreg_128_align2 = COPY %7
+ %11.sub1:vreg_128_align2 = COPY %8
+ %11.sub2:vreg_128_align2 = COPY %9
+ %11.sub3:vreg_128_align2 = COPY %10
+ %12:vgpr_32 = V_ADD_U32_e32 9, %1, implicit $exec
+ %13:vgpr_32 = V_ADD_U32_e32 10, %1, implicit $exec
+ %14:vgpr_32 = V_ADD_U32_e32 11, %1, implicit $exec
+ %15:vgpr_32 = V_ADD_U32_e32 12, %1, implicit $exec
+ undef %16.sub0:vreg_128_align2 = COPY %12
+ %16.sub1:vreg_128_align2 = COPY %13
+ %16.sub2:vreg_128_align2 = COPY %14
+ %16.sub3:vreg_128_align2 = COPY %15
+ %17:vgpr_32 = V_ADD_U32_e32 13, %1, implicit $exec
+ %18:vgpr_32 = V_ADD_U32_e32 14, %1, implicit $exec
+ %19:vgpr_32 = V_ADD_U32_e32 15, %1, implicit $exec
+ %20:vgpr_32 = V_ADD_U32_e32 16, %1, implicit $exec
+ undef %21.sub0:vreg_128_align2 = COPY %17
+ %21.sub1:vreg_128_align2 = COPY %18
+ %21.sub2:vreg_128_align2 = COPY %19
+ %21.sub3:vreg_128_align2 = COPY %20
+ S_CMP_EQ_U32 $sgpr7, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.2, implicit $scc
+ S_BRANCH %bb.1
+ bb.1:
+ successors: %bb.1(0x78000000), %bb.3(0x08000000)
+ liveins: $sgpr7, $sgpr6
+ %6.sub0:vreg_128_align2 = V_ADD_U32_e32 1, %6.sub0, implicit $exec
+ %6.sub1:vreg_128_align2 = V_ADD_U32_e32 2, %6.sub1, implicit $exec
+ %6.sub2:vreg_128_align2 = V_ADD_U32_e32 3, %6.sub2, implicit $exec
+ %6.sub3:vreg_128_align2 = V_ADD_U32_e32 4, %6.sub3, implicit $exec
+ %11.sub0:vreg_128_align2 = V_ADD_U32_e32 2, %11.sub0, implicit $exec
+ %11.sub1:vreg_128_align2 = V_ADD_U32_e32 3, %11.sub1, implicit $exec
+ %11.sub2:vreg_128_align2 = V_ADD_U32_e32 4, %11.sub2, implicit $exec
+ %11.sub3:vreg_128_align2 = V_ADD_U32_e32 5, %11.sub3, implicit $exec
+ %16.sub0:vreg_128_align2 = V_ADD_U32_e32 3, %16.sub0, implicit $exec
+ %16.sub1:vreg_128_align2 = V_ADD_U32_e32 4, %16.sub1, implicit $exec
+ %16.sub2:vreg_128_align2 = V_ADD_U32_e32 5, %16.sub2, implicit $exec
+ %16.sub3:vreg_128_align2 = V_ADD_U32_e32 6, %16.sub3, implicit $exec
+ %21.sub0:vreg_128_align2 = V_ADD_U32_e32 4, %21.sub0, implicit $exec
+ %21.sub1:vreg_128_align2 = V_ADD_U32_e32 5, %21.sub1, implicit $exec
+ %21.sub2:vreg_128_align2 = V_ADD_U32_e32 6, %21.sub2, implicit $exec
+ %21.sub3:vreg_128_align2 = V_ADD_U32_e32 7, %21.sub3, implicit $exec
+ $sgpr7 = S_SUB_U32 $sgpr7, 1, implicit-def $scc
+ S_CMP_LG_U32 $sgpr7, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.1, implicit $scc
+ S_BRANCH %bb.3
+ bb.2:
+ successors: %bb.4
+ liveins: $sgpr6
+ S_BRANCH %bb.4
+ bb.3:
+ successors: %bb.4
+ liveins: $sgpr6
+ S_BRANCH %bb.4
+ bb.4:
+ liveins: $sgpr6
+ successors: %bb.5
+ S_BRANCH %bb.5
+ bb.5:
+ successors: %bb.5(0x78000000), %bb.6(0x08000000)
+ liveins: $sgpr6
+ %22:vgpr_32 = GLOBAL_LOAD_DWORD %0, 0, 0, implicit $exec :: (load (s32), addrspace 1)
+ %23:vgpr_32 = V_ADD_U32_e32 %22, %6.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %23, 0, 0, implicit $exec :: (store (s32), addrspace 1)
+ %24:vgpr_32 = GLOBAL_LOAD_DWORD %0, 4, 0, implicit $exec :: (load (s32), addrspace 1)
+ %25:vgpr_32 = V_ADD_U32_e32 %24, %6.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %25, 4, 0, implicit $exec :: (store (s32), addrspace 1)
+ %26:vgpr_32 = GLOBAL_LOAD_DWORD %0, 8, 0, implicit $exec :: (load (s32), addrspace 1)
+ %27:vgpr_32 = V_ADD_U32_e32 %26, %6.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %27, 8, 0, implicit $exec :: (store (s32), addrspace 1)
+ %28:vgpr_32 = GLOBAL_LOAD_DWORD %0, 12, 0, implicit $exec :: (load (s32), addrspace 1)
+ %29:vgpr_32 = V_ADD_U32_e32 %28, %6.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %29, 12, 0, implicit $exec :: (store (s32), addrspace 1)
+ %30:vgpr_32 = GLOBAL_LOAD_DWORD %0, 16, 0, implicit $exec :: (load (s32), addrspace 1)
+ %31:vgpr_32 = V_ADD_U32_e32 %30, %11.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %31, 16, 0, implicit $exec :: (store (s32), addrspace 1)
+ %32:vgpr_32 = GLOBAL_LOAD_DWORD %0, 20, 0, implicit $exec :: (load (s32), addrspace 1)
+ %33:vgpr_32 = V_ADD_U32_e32 %32, %11.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %33, 20, 0, implicit $exec :: (store (s32), addrspace 1)
+ %34:vgpr_32 = GLOBAL_LOAD_DWORD %0, 24, 0, implicit $exec :: (load (s32), addrspace 1)
+ %35:vgpr_32 = V_ADD_U32_e32 %34, %11.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %35, 24, 0, implicit $exec :: (store (s32), addrspace 1)
+ %36:vgpr_32 = GLOBAL_LOAD_DWORD %0, 28, 0, implicit $exec :: (load (s32), addrspace 1)
+ %37:vgpr_32 = V_ADD_U32_e32 %36, %11.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %37, 28, 0, implicit $exec :: (store (s32), addrspace 1)
+ %38:vgpr_32 = GLOBAL_LOAD_DWORD %0, 32, 0, implicit $exec :: (load (s32), addrspace 1)
+ %39:vgpr_32 = V_ADD_U32_e32 %38, %16.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %39, 32, 0, implicit $exec :: (store (s32), addrspace 1)
+ %40:vgpr_32 = GLOBAL_LOAD_DWORD %0, 36, 0, implicit $exec :: (load (s32), addrspace 1)
+ %41:vgpr_32 = V_ADD_U32_e32 %40, %16.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %41, 36, 0, implicit $exec :: (store (s32), addrspace 1)
+ %42:vgpr_32 = GLOBAL_LOAD_DWORD %0, 40, 0, implicit $exec :: (load (s32), addrspace 1)
+ %43:vgpr_32 = V_ADD_U32_e32 %42, %16.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %43, 40, 0, implicit $exec :: (store (s32), addrspace 1)
+ %44:vgpr_32 = GLOBAL_LOAD_DWORD %0, 44, 0, implicit $exec :: (load (s32), addrspace 1)
+ %45:vgpr_32 = V_ADD_U32_e32 %44, %16.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %45, 44, 0, implicit $exec :: (store (s32), addrspace 1)
+ %46:vgpr_32 = GLOBAL_LOAD_DWORD %0, 48, 0, implicit $exec :: (load (s32), addrspace 1)
+ %47:vgpr_32 = V_ADD_U32_e32 %46, %21.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %47, 48, 0, implicit $exec :: (store (s32), addrspace 1)
+ %48:vgpr_32 = GLOBAL_LOAD_DWORD %0, 52, 0, implicit $exec :: (load (s32), addrspace 1)
+ %49:vgpr_32 = V_ADD_U32_e32 %48, %21.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %49, 52, 0, implicit $exec :: (store (s32), addrspace 1)
+ %50:vgpr_32 = GLOBAL_LOAD_DWORD %0, 56, 0, implicit $exec :: (load (s32), addrspace 1)
+ %51:vgpr_32 = V_ADD_U32_e32 %50, %21.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %51, 56, 0, implicit $exec :: (store (s32), addrspace 1)
+ %52:vgpr_32 = GLOBAL_LOAD_DWORD %0, 60, 0, implicit $exec :: (load (s32), addrspace 1)
+ %53:vgpr_32 = V_ADD_U32_e32 %52, %21.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %53, 60, 0, implicit $exec :: (store (s32), addrspace 1)
+ $sgpr6 = S_SUB_U32 $sgpr6, 1, implicit-def $scc
+ S_CMP_LG_U32 $sgpr6, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.5, implicit $scc
+ S_BRANCH %bb.6
+ bb.6:
+ S_ENDPGM 0
+...
+
+# An intervening store prevents proving redundancy on bb.2. Keep its reload,
+# move the join store to that path, and remove the reload from bb.3.
+# CHECK-LABEL: name: partial_join
+# CHECK: bb.2:
+# CHECK: [[P:%[0-9]+]]:av_128_align2 = SI_SPILL_AV128_RESTORE [[PS:%stack.[0-9]+]],
+# CHECK-NEXT: [[P2:%[0-9]+]]:av_128_align2 = SI_SPILL_AV128_RESTORE [[PS2:%stack.[0-9]+]],
+# CHECK-NEXT: SI_SPILL_AV128_SAVE $vgpr10_vgpr11_vgpr12_vgpr13,
+# ON-NEXT: SI_SPILL_AV128_SAVE [[P]], [[PS]],
+# ON-NEXT: SI_SPILL_AV128_SAVE [[P2]], [[PS2]],
+# CHECK-NEXT: S_BRANCH %bb.4
+# CHECK: bb.3:
+# CHECK: SI_SPILL_AV128_SAVE
+# OFF-NEXT: [[P]]:av_128_align2 = SI_SPILL_AV128_RESTORE [[PS]],
+# OFF-NEXT: [[P2]]:av_128_align2 = SI_SPILL_AV128_RESTORE [[PS2]],
+# CHECK-NEXT: S_BRANCH %bb.4
+# CHECK: bb.4:
+# OFF: SI_SPILL_AV128_SAVE [[P]], [[PS]],
+# OFF-NEXT: SI_SPILL_AV128_SAVE [[P2]], [[PS2]],
+# ON-NOT: SI_SPILL_AV128_SAVE
+# CHECK: SI_SPILL_AV64_RESTORE
+# CHECK-NEXT: S_BRANCH %bb.5
+
+---
+name: partial_join
+tracksRegLiveness: true
+machineFunctionInfo:
+ stackPtrOffsetReg: '$sgpr32'
+body: |
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $vgpr0_vgpr1, $vgpr2, $sgpr7, $vgpr10_vgpr11_vgpr12_vgpr13
+ $sgpr6 = S_MOV_B32 4
+ %0:vreg_64_align2 = COPY $vgpr0_vgpr1
+ %1:vgpr_32 = COPY $vgpr2
+ %2:vgpr_32 = V_ADD_U32_e32 1, %1, implicit $exec
+ %3:vgpr_32 = V_ADD_U32_e32 2, %1, implicit $exec
+ %4:vgpr_32 = V_ADD_U32_e32 3, %1, implicit $exec
+ %5:vgpr_32 = V_ADD_U32_e32 4, %1, implicit $exec
+ undef %6.sub0:vreg_128_align2 = COPY %2
+ %6.sub1:vreg_128_align2 = COPY %3
+ %6.sub2:vreg_128_align2 = COPY %4
+ %6.sub3:vreg_128_align2 = COPY %5
+ %7:vgpr_32 = V_ADD_U32_e32 5, %1, implicit $exec
+ %8:vgpr_32 = V_ADD_U32_e32 6, %1, implicit $exec
+ %9:vgpr_32 = V_ADD_U32_e32 7, %1, implicit $exec
+ %10:vgpr_32 = V_ADD_U32_e32 8, %1, implicit $exec
+ undef %11.sub0:vreg_128_align2 = COPY %7
+ %11.sub1:vreg_128_align2 = COPY %8
+ %11.sub2:vreg_128_align2 = COPY %9
+ %11.sub3:vreg_128_align2 = COPY %10
+ %12:vgpr_32 = V_ADD_U32_e32 9, %1, implicit $exec
+ %13:vgpr_32 = V_ADD_U32_e32 10, %1, implicit $exec
+ %14:vgpr_32 = V_ADD_U32_e32 11, %1, implicit $exec
+ %15:vgpr_32 = V_ADD_U32_e32 12, %1, implicit $exec
+ undef %16.sub0:vreg_128_align2 = COPY %12
+ %16.sub1:vreg_128_align2 = COPY %13
+ %16.sub2:vreg_128_align2 = COPY %14
+ %16.sub3:vreg_128_align2 = COPY %15
+ %17:vgpr_32 = V_ADD_U32_e32 13, %1, implicit $exec
+ %18:vgpr_32 = V_ADD_U32_e32 14, %1, implicit $exec
+ %19:vgpr_32 = V_ADD_U32_e32 15, %1, implicit $exec
+ %20:vgpr_32 = V_ADD_U32_e32 16, %1, implicit $exec
+ undef %21.sub0:vreg_128_align2 = COPY %17
+ %21.sub1:vreg_128_align2 = COPY %18
+ %21.sub2:vreg_128_align2 = COPY %19
+ %21.sub3:vreg_128_align2 = COPY %20
+ S_CMP_EQ_U32 $sgpr7, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.2, implicit $scc
+ S_BRANCH %bb.1
+ bb.1:
+ successors: %bb.1(0x78000000), %bb.3(0x08000000)
+ liveins: $sgpr7, $sgpr6
+ %6.sub0:vreg_128_align2 = V_ADD_U32_e32 1, %6.sub0, implicit $exec
+ %6.sub1:vreg_128_align2 = V_ADD_U32_e32 2, %6.sub1, implicit $exec
+ %6.sub2:vreg_128_align2 = V_ADD_U32_e32 3, %6.sub2, implicit $exec
+ %6.sub3:vreg_128_align2 = V_ADD_U32_e32 4, %6.sub3, implicit $exec
+ %11.sub0:vreg_128_align2 = V_ADD_U32_e32 2, %11.sub0, implicit $exec
+ %11.sub1:vreg_128_align2 = V_ADD_U32_e32 3, %11.sub1, implicit $exec
+ %11.sub2:vreg_128_align2 = V_ADD_U32_e32 4, %11.sub2, implicit $exec
+ %11.sub3:vreg_128_align2 = V_ADD_U32_e32 5, %11.sub3, implicit $exec
+ %16.sub0:vreg_128_align2 = V_ADD_U32_e32 3, %16.sub0, implicit $exec
+ %16.sub1:vreg_128_align2 = V_ADD_U32_e32 4, %16.sub1, implicit $exec
+ %16.sub2:vreg_128_align2 = V_ADD_U32_e32 5, %16.sub2, implicit $exec
+ %16.sub3:vreg_128_align2 = V_ADD_U32_e32 6, %16.sub3, implicit $exec
+ %21.sub0:vreg_128_align2 = V_ADD_U32_e32 4, %21.sub0, implicit $exec
+ %21.sub1:vreg_128_align2 = V_ADD_U32_e32 5, %21.sub1, implicit $exec
+ %21.sub2:vreg_128_align2 = V_ADD_U32_e32 6, %21.sub2, implicit $exec
+ %21.sub3:vreg_128_align2 = V_ADD_U32_e32 7, %21.sub3, implicit $exec
+ $sgpr7 = S_SUB_U32 $sgpr7, 1, implicit-def $scc
+ S_CMP_LG_U32 $sgpr7, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.1, implicit $scc
+ S_BRANCH %bb.3
+ bb.2:
+ successors: %bb.4
+ liveins: $sgpr6, $vgpr10_vgpr11_vgpr12_vgpr13
+ %11:vreg_128_align2 = COPY $vgpr10_vgpr11_vgpr12_vgpr13
+ S_BRANCH %bb.4
+ bb.3:
+ successors: %bb.4
+ liveins: $sgpr6
+ S_BRANCH %bb.4
+ bb.4:
+ liveins: $sgpr6
+ successors: %bb.5
+ S_BRANCH %bb.5
+ bb.5:
+ successors: %bb.5(0x78000000), %bb.6(0x08000000)
+ liveins: $sgpr6
+ %22:vgpr_32 = GLOBAL_LOAD_DWORD %0, 0, 0, implicit $exec :: (load (s32), addrspace 1)
+ %23:vgpr_32 = V_ADD_U32_e32 %22, %6.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %23, 0, 0, implicit $exec :: (store (s32), addrspace 1)
+ %24:vgpr_32 = GLOBAL_LOAD_DWORD %0, 4, 0, implicit $exec :: (load (s32), addrspace 1)
+ %25:vgpr_32 = V_ADD_U32_e32 %24, %6.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %25, 4, 0, implicit $exec :: (store (s32), addrspace 1)
+ %26:vgpr_32 = GLOBAL_LOAD_DWORD %0, 8, 0, implicit $exec :: (load (s32), addrspace 1)
+ %27:vgpr_32 = V_ADD_U32_e32 %26, %6.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %27, 8, 0, implicit $exec :: (store (s32), addrspace 1)
+ %28:vgpr_32 = GLOBAL_LOAD_DWORD %0, 12, 0, implicit $exec :: (load (s32), addrspace 1)
+ %29:vgpr_32 = V_ADD_U32_e32 %28, %6.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %29, 12, 0, implicit $exec :: (store (s32), addrspace 1)
+ %30:vgpr_32 = GLOBAL_LOAD_DWORD %0, 16, 0, implicit $exec :: (load (s32), addrspace 1)
+ %31:vgpr_32 = V_ADD_U32_e32 %30, %11.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %31, 16, 0, implicit $exec :: (store (s32), addrspace 1)
+ %32:vgpr_32 = GLOBAL_LOAD_DWORD %0, 20, 0, implicit $exec :: (load (s32), addrspace 1)
+ %33:vgpr_32 = V_ADD_U32_e32 %32, %11.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %33, 20, 0, implicit $exec :: (store (s32), addrspace 1)
+ %34:vgpr_32 = GLOBAL_LOAD_DWORD %0, 24, 0, implicit $exec :: (load (s32), addrspace 1)
+ %35:vgpr_32 = V_ADD_U32_e32 %34, %11.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %35, 24, 0, implicit $exec :: (store (s32), addrspace 1)
+ %36:vgpr_32 = GLOBAL_LOAD_DWORD %0, 28, 0, implicit $exec :: (load (s32), addrspace 1)
+ %37:vgpr_32 = V_ADD_U32_e32 %36, %11.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %37, 28, 0, implicit $exec :: (store (s32), addrspace 1)
+ %38:vgpr_32 = GLOBAL_LOAD_DWORD %0, 32, 0, implicit $exec :: (load (s32), addrspace 1)
+ %39:vgpr_32 = V_ADD_U32_e32 %38, %16.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %39, 32, 0, implicit $exec :: (store (s32), addrspace 1)
+ %40:vgpr_32 = GLOBAL_LOAD_DWORD %0, 36, 0, implicit $exec :: (load (s32), addrspace 1)
+ %41:vgpr_32 = V_ADD_U32_e32 %40, %16.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %41, 36, 0, implicit $exec :: (store (s32), addrspace 1)
+ %42:vgpr_32 = GLOBAL_LOAD_DWORD %0, 40, 0, implicit $exec :: (load (s32), addrspace 1)
+ %43:vgpr_32 = V_ADD_U32_e32 %42, %16.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %43, 40, 0, implicit $exec :: (store (s32), addrspace 1)
+ %44:vgpr_32 = GLOBAL_LOAD_DWORD %0, 44, 0, implicit $exec :: (load (s32), addrspace 1)
+ %45:vgpr_32 = V_ADD_U32_e32 %44, %16.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %45, 44, 0, implicit $exec :: (store (s32), addrspace 1)
+ %46:vgpr_32 = GLOBAL_LOAD_DWORD %0, 48, 0, implicit $exec :: (load (s32), addrspace 1)
+ %47:vgpr_32 = V_ADD_U32_e32 %46, %21.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %47, 48, 0, implicit $exec :: (store (s32), addrspace 1)
+ %48:vgpr_32 = GLOBAL_LOAD_DWORD %0, 52, 0, implicit $exec :: (load (s32), addrspace 1)
+ %49:vgpr_32 = V_ADD_U32_e32 %48, %21.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %49, 52, 0, implicit $exec :: (store (s32), addrspace 1)
+ %50:vgpr_32 = GLOBAL_LOAD_DWORD %0, 56, 0, implicit $exec :: (load (s32), addrspace 1)
+ %51:vgpr_32 = V_ADD_U32_e32 %50, %21.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %51, 56, 0, implicit $exec :: (store (s32), addrspace 1)
+ %52:vgpr_32 = GLOBAL_LOAD_DWORD %0, 60, 0, implicit $exec :: (load (s32), addrspace 1)
+ %53:vgpr_32 = V_ADD_U32_e32 %52, %21.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %53, 60, 0, implicit $exec :: (store (s32), addrspace 1)
+ $sgpr6 = S_SUB_U32 $sgpr6, 1, implicit-def $scc
+ S_CMP_LG_U32 $sgpr6, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.5, implicit $scc
+ S_BRANCH %bb.6
+ bb.6:
+ S_ENDPGM 0
+...
+
+# The terminator changes EXEC. The reload and store use different masks.
+# CHECK-LABEL: name: mask_join
+# CHECK: bb.2:
+# CHECK: [[M:%[0-9]+]]:av_128_align2 = SI_SPILL_AV128_RESTORE [[MS:%stack.[0-9]+]],
+# CHECK-NEXT: S_BRANCH %bb.4
+# CHECK: bb.3:
+# CHECK: [[M]]:av_128_align2 = SI_SPILL_AV128_RESTORE [[MS]],
+# CHECK-NEXT: S_BRANCH %bb.4, implicit-def $exec
+# CHECK: bb.4:
+# CHECK: SI_SPILL_AV128_SAVE [[M]], [[MS]],
+# CHECK-NEXT: {{%[0-9]+}}:vreg_64_align2 = SI_SPILL_AV64_RESTORE
+
+---
+name: mask_join
+tracksRegLiveness: true
+machineFunctionInfo:
+ stackPtrOffsetReg: '$sgpr32'
+body: |
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $vgpr0_vgpr1, $vgpr2, $sgpr7
+ $sgpr6 = S_MOV_B32 4
+ %0:vreg_64_align2 = COPY $vgpr0_vgpr1
+ %1:vgpr_32 = COPY $vgpr2
+ %2:vgpr_32 = V_ADD_U32_e32 1, %1, implicit $exec
+ %3:vgpr_32 = V_ADD_U32_e32 2, %1, implicit $exec
+ %4:vgpr_32 = V_ADD_U32_e32 3, %1, implicit $exec
+ %5:vgpr_32 = V_ADD_U32_e32 4, %1, implicit $exec
+ undef %6.sub0:vreg_128_align2 = COPY %2
+ %6.sub1:vreg_128_align2 = COPY %3
+ %6.sub2:vreg_128_align2 = COPY %4
+ %6.sub3:vreg_128_align2 = COPY %5
+ %7:vgpr_32 = V_ADD_U32_e32 5, %1, implicit $exec
+ %8:vgpr_32 = V_ADD_U32_e32 6, %1, implicit $exec
+ %9:vgpr_32 = V_ADD_U32_e32 7, %1, implicit $exec
+ %10:vgpr_32 = V_ADD_U32_e32 8, %1, implicit $exec
+ undef %11.sub0:vreg_128_align2 = COPY %7
+ %11.sub1:vreg_128_align2 = COPY %8
+ %11.sub2:vreg_128_align2 = COPY %9
+ %11.sub3:vreg_128_align2 = COPY %10
+ %12:vgpr_32 = V_ADD_U32_e32 9, %1, implicit $exec
+ %13:vgpr_32 = V_ADD_U32_e32 10, %1, implicit $exec
+ %14:vgpr_32 = V_ADD_U32_e32 11, %1, implicit $exec
+ %15:vgpr_32 = V_ADD_U32_e32 12, %1, implicit $exec
+ undef %16.sub0:vreg_128_align2 = COPY %12
+ %16.sub1:vreg_128_align2 = COPY %13
+ %16.sub2:vreg_128_align2 = COPY %14
+ %16.sub3:vreg_128_align2 = COPY %15
+ %17:vgpr_32 = V_ADD_U32_e32 13, %1, implicit $exec
+ %18:vgpr_32 = V_ADD_U32_e32 14, %1, implicit $exec
+ %19:vgpr_32 = V_ADD_U32_e32 15, %1, implicit $exec
+ %20:vgpr_32 = V_ADD_U32_e32 16, %1, implicit $exec
+ undef %21.sub0:vreg_128_align2 = COPY %17
+ %21.sub1:vreg_128_align2 = COPY %18
+ %21.sub2:vreg_128_align2 = COPY %19
+ %21.sub3:vreg_128_align2 = COPY %20
+ S_CMP_EQ_U32 $sgpr7, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.2, implicit $scc
+ S_BRANCH %bb.1
+ bb.1:
+ successors: %bb.1(0x78000000), %bb.3(0x08000000)
+ liveins: $sgpr7, $sgpr6
+ %6.sub0:vreg_128_align2 = V_ADD_U32_e32 1, %6.sub0, implicit $exec
+ %6.sub1:vreg_128_align2 = V_ADD_U32_e32 2, %6.sub1, implicit $exec
+ %6.sub2:vreg_128_align2 = V_ADD_U32_e32 3, %6.sub2, implicit $exec
+ %6.sub3:vreg_128_align2 = V_ADD_U32_e32 4, %6.sub3, implicit $exec
+ %11.sub0:vreg_128_align2 = V_ADD_U32_e32 2, %11.sub0, implicit $exec
+ %11.sub1:vreg_128_align2 = V_ADD_U32_e32 3, %11.sub1, implicit $exec
+ %11.sub2:vreg_128_align2 = V_ADD_U32_e32 4, %11.sub2, implicit $exec
+ %11.sub3:vreg_128_align2 = V_ADD_U32_e32 5, %11.sub3, implicit $exec
+ %16.sub0:vreg_128_align2 = V_ADD_U32_e32 3, %16.sub0, implicit $exec
+ %16.sub1:vreg_128_align2 = V_ADD_U32_e32 4, %16.sub1, implicit $exec
+ %16.sub2:vreg_128_align2 = V_ADD_U32_e32 5, %16.sub2, implicit $exec
+ %16.sub3:vreg_128_align2 = V_ADD_U32_e32 6, %16.sub3, implicit $exec
+ %21.sub0:vreg_128_align2 = V_ADD_U32_e32 4, %21.sub0, implicit $exec
+ %21.sub1:vreg_128_align2 = V_ADD_U32_e32 5, %21.sub1, implicit $exec
+ %21.sub2:vreg_128_align2 = V_ADD_U32_e32 6, %21.sub2, implicit $exec
+ %21.sub3:vreg_128_align2 = V_ADD_U32_e32 7, %21.sub3, implicit $exec
+ $sgpr7 = S_SUB_U32 $sgpr7, 1, implicit-def $scc
+ S_CMP_LG_U32 $sgpr7, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.1, implicit $scc
+ S_BRANCH %bb.3
+ bb.2:
+ successors: %bb.4
+ liveins: $sgpr6
+ S_BRANCH %bb.4
+ bb.3:
+ successors: %bb.4
+ liveins: $sgpr6
+ S_BRANCH %bb.4, implicit-def $exec
+ bb.4:
+ liveins: $sgpr6
+ successors: %bb.5
+ S_BRANCH %bb.5
+ bb.5:
+ successors: %bb.5(0x78000000), %bb.6(0x08000000)
+ liveins: $sgpr6
+ %22:vgpr_32 = GLOBAL_LOAD_DWORD %0, 0, 0, implicit $exec :: (load (s32), addrspace 1)
+ %23:vgpr_32 = V_ADD_U32_e32 %22, %6.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %23, 0, 0, implicit $exec :: (store (s32), addrspace 1)
+ %24:vgpr_32 = GLOBAL_LOAD_DWORD %0, 4, 0, implicit $exec :: (load (s32), addrspace 1)
+ %25:vgpr_32 = V_ADD_U32_e32 %24, %6.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %25, 4, 0, implicit $exec :: (store (s32), addrspace 1)
+ %26:vgpr_32 = GLOBAL_LOAD_DWORD %0, 8, 0, implicit $exec :: (load (s32), addrspace 1)
+ %27:vgpr_32 = V_ADD_U32_e32 %26, %6.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %27, 8, 0, implicit $exec :: (store (s32), addrspace 1)
+ %28:vgpr_32 = GLOBAL_LOAD_DWORD %0, 12, 0, implicit $exec :: (load (s32), addrspace 1)
+ %29:vgpr_32 = V_ADD_U32_e32 %28, %6.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %29, 12, 0, implicit $exec :: (store (s32), addrspace 1)
+ %30:vgpr_32 = GLOBAL_LOAD_DWORD %0, 16, 0, implicit $exec :: (load (s32), addrspace 1)
+ %31:vgpr_32 = V_ADD_U32_e32 %30, %11.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %31, 16, 0, implicit $exec :: (store (s32), addrspace 1)
+ %32:vgpr_32 = GLOBAL_LOAD_DWORD %0, 20, 0, implicit $exec :: (load (s32), addrspace 1)
+ %33:vgpr_32 = V_ADD_U32_e32 %32, %11.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %33, 20, 0, implicit $exec :: (store (s32), addrspace 1)
+ %34:vgpr_32 = GLOBAL_LOAD_DWORD %0, 24, 0, implicit $exec :: (load (s32), addrspace 1)
+ %35:vgpr_32 = V_ADD_U32_e32 %34, %11.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %35, 24, 0, implicit $exec :: (store (s32), addrspace 1)
+ %36:vgpr_32 = GLOBAL_LOAD_DWORD %0, 28, 0, implicit $exec :: (load (s32), addrspace 1)
+ %37:vgpr_32 = V_ADD_U32_e32 %36, %11.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %37, 28, 0, implicit $exec :: (store (s32), addrspace 1)
+ %38:vgpr_32 = GLOBAL_LOAD_DWORD %0, 32, 0, implicit $exec :: (load (s32), addrspace 1)
+ %39:vgpr_32 = V_ADD_U32_e32 %38, %16.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %39, 32, 0, implicit $exec :: (store (s32), addrspace 1)
+ %40:vgpr_32 = GLOBAL_LOAD_DWORD %0, 36, 0, implicit $exec :: (load (s32), addrspace 1)
+ %41:vgpr_32 = V_ADD_U32_e32 %40, %16.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %41, 36, 0, implicit $exec :: (store (s32), addrspace 1)
+ %42:vgpr_32 = GLOBAL_LOAD_DWORD %0, 40, 0, implicit $exec :: (load (s32), addrspace 1)
+ %43:vgpr_32 = V_ADD_U32_e32 %42, %16.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %43, 40, 0, implicit $exec :: (store (s32), addrspace 1)
+ %44:vgpr_32 = GLOBAL_LOAD_DWORD %0, 44, 0, implicit $exec :: (load (s32), addrspace 1)
+ %45:vgpr_32 = V_ADD_U32_e32 %44, %16.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %45, 44, 0, implicit $exec :: (store (s32), addrspace 1)
+ %46:vgpr_32 = GLOBAL_LOAD_DWORD %0, 48, 0, implicit $exec :: (load (s32), addrspace 1)
+ %47:vgpr_32 = V_ADD_U32_e32 %46, %21.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %47, 48, 0, implicit $exec :: (store (s32), addrspace 1)
+ %48:vgpr_32 = GLOBAL_LOAD_DWORD %0, 52, 0, implicit $exec :: (load (s32), addrspace 1)
+ %49:vgpr_32 = V_ADD_U32_e32 %48, %21.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %49, 52, 0, implicit $exec :: (store (s32), addrspace 1)
+ %50:vgpr_32 = GLOBAL_LOAD_DWORD %0, 56, 0, implicit $exec :: (load (s32), addrspace 1)
+ %51:vgpr_32 = V_ADD_U32_e32 %50, %21.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %51, 56, 0, implicit $exec :: (store (s32), addrspace 1)
+ %52:vgpr_32 = GLOBAL_LOAD_DWORD %0, 60, 0, implicit $exec :: (load (s32), addrspace 1)
+ %53:vgpr_32 = V_ADD_U32_e32 %52, %21.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %53, 60, 0, implicit $exec :: (store (s32), addrspace 1)
+ $sgpr6 = S_SUB_U32 $sgpr6, 1, implicit-def $scc
+ S_CMP_LG_U32 $sgpr6, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.5, implicit $scc
+ S_BRANCH %bb.6
+ bb.6:
+ S_ENDPGM 0
+...
+
+# Unknown side effects on both incoming paths prevent removing the join store.
+# CHECK-LABEL: name: barrier_join
+# CHECK: bb.2:
+# CHECK: [[B:%[0-9]+]]:av_128_align2 = SI_SPILL_AV128_RESTORE [[BS:%stack.[0-9]+]],
+# CHECK: INLINEASM
+# CHECK-NEXT: S_BRANCH %bb.4
+# CHECK: bb.3:
+# CHECK: [[B]]:av_128_align2 = SI_SPILL_AV128_RESTORE [[BS]],
+# CHECK: INLINEASM
+# CHECK-NEXT: S_BRANCH %bb.4
+# CHECK: bb.4:
+# CHECK: SI_SPILL_AV128_SAVE [[B]], [[BS]],
+# CHECK-NEXT: {{%[0-9]+}}:vreg_64_align2 = SI_SPILL_AV64_RESTORE
+
+---
+name: barrier_join
+tracksRegLiveness: true
+machineFunctionInfo:
+ stackPtrOffsetReg: '$sgpr32'
+body: |
+ bb.0:
+ successors: %bb.1, %bb.2
+ liveins: $vgpr0_vgpr1, $vgpr2, $sgpr7
+ $sgpr6 = S_MOV_B32 4
+ %0:vreg_64_align2 = COPY $vgpr0_vgpr1
+ %1:vgpr_32 = COPY $vgpr2
+ %2:vgpr_32 = V_ADD_U32_e32 1, %1, implicit $exec
+ %3:vgpr_32 = V_ADD_U32_e32 2, %1, implicit $exec
+ %4:vgpr_32 = V_ADD_U32_e32 3, %1, implicit $exec
+ %5:vgpr_32 = V_ADD_U32_e32 4, %1, implicit $exec
+ undef %6.sub0:vreg_128_align2 = COPY %2
+ %6.sub1:vreg_128_align2 = COPY %3
+ %6.sub2:vreg_128_align2 = COPY %4
+ %6.sub3:vreg_128_align2 = COPY %5
+ %7:vgpr_32 = V_ADD_U32_e32 5, %1, implicit $exec
+ %8:vgpr_32 = V_ADD_U32_e32 6, %1, implicit $exec
+ %9:vgpr_32 = V_ADD_U32_e32 7, %1, implicit $exec
+ %10:vgpr_32 = V_ADD_U32_e32 8, %1, implicit $exec
+ undef %11.sub0:vreg_128_align2 = COPY %7
+ %11.sub1:vreg_128_align2 = COPY %8
+ %11.sub2:vreg_128_align2 = COPY %9
+ %11.sub3:vreg_128_align2 = COPY %10
+ %12:vgpr_32 = V_ADD_U32_e32 9, %1, implicit $exec
+ %13:vgpr_32 = V_ADD_U32_e32 10, %1, implicit $exec
+ %14:vgpr_32 = V_ADD_U32_e32 11, %1, implicit $exec
+ %15:vgpr_32 = V_ADD_U32_e32 12, %1, implicit $exec
+ undef %16.sub0:vreg_128_align2 = COPY %12
+ %16.sub1:vreg_128_align2 = COPY %13
+ %16.sub2:vreg_128_align2 = COPY %14
+ %16.sub3:vreg_128_align2 = COPY %15
+ %17:vgpr_32 = V_ADD_U32_e32 13, %1, implicit $exec
+ %18:vgpr_32 = V_ADD_U32_e32 14, %1, implicit $exec
+ %19:vgpr_32 = V_ADD_U32_e32 15, %1, implicit $exec
+ %20:vgpr_32 = V_ADD_U32_e32 16, %1, implicit $exec
+ undef %21.sub0:vreg_128_align2 = COPY %17
+ %21.sub1:vreg_128_align2 = COPY %18
+ %21.sub2:vreg_128_align2 = COPY %19
+ %21.sub3:vreg_128_align2 = COPY %20
+ S_CMP_EQ_U32 $sgpr7, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.2, implicit $scc
+ S_BRANCH %bb.1
+ bb.1:
+ successors: %bb.1(0x78000000), %bb.3(0x08000000)
+ liveins: $sgpr7, $sgpr6
+ %6.sub0:vreg_128_align2 = V_ADD_U32_e32 1, %6.sub0, implicit $exec
+ %6.sub1:vreg_128_align2 = V_ADD_U32_e32 2, %6.sub1, implicit $exec
+ %6.sub2:vreg_128_align2 = V_ADD_U32_e32 3, %6.sub2, implicit $exec
+ %6.sub3:vreg_128_align2 = V_ADD_U32_e32 4, %6.sub3, implicit $exec
+ %11.sub0:vreg_128_align2 = V_ADD_U32_e32 2, %11.sub0, implicit $exec
+ %11.sub1:vreg_128_align2 = V_ADD_U32_e32 3, %11.sub1, implicit $exec
+ %11.sub2:vreg_128_align2 = V_ADD_U32_e32 4, %11.sub2, implicit $exec
+ %11.sub3:vreg_128_align2 = V_ADD_U32_e32 5, %11.sub3, implicit $exec
+ %16.sub0:vreg_128_align2 = V_ADD_U32_e32 3, %16.sub0, implicit $exec
+ %16.sub1:vreg_128_align2 = V_ADD_U32_e32 4, %16.sub1, implicit $exec
+ %16.sub2:vreg_128_align2 = V_ADD_U32_e32 5, %16.sub2, implicit $exec
+ %16.sub3:vreg_128_align2 = V_ADD_U32_e32 6, %16.sub3, implicit $exec
+ %21.sub0:vreg_128_align2 = V_ADD_U32_e32 4, %21.sub0, implicit $exec
+ %21.sub1:vreg_128_align2 = V_ADD_U32_e32 5, %21.sub1, implicit $exec
+ %21.sub2:vreg_128_align2 = V_ADD_U32_e32 6, %21.sub2, implicit $exec
+ %21.sub3:vreg_128_align2 = V_ADD_U32_e32 7, %21.sub3, implicit $exec
+ $sgpr7 = S_SUB_U32 $sgpr7, 1, implicit-def $scc
+ S_CMP_LG_U32 $sgpr7, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.1, implicit $scc
+ S_BRANCH %bb.3
+ bb.2:
+ successors: %bb.4
+ liveins: $sgpr6
+ INLINEASM &"", sideeffect attdialect, implicit %6:vreg_128_align2
+ S_BRANCH %bb.4
+ bb.3:
+ successors: %bb.4
+ liveins: $sgpr6
+ INLINEASM &"", sideeffect attdialect, implicit %6:vreg_128_align2
+ S_BRANCH %bb.4
+ bb.4:
+ liveins: $sgpr6
+ successors: %bb.5
+ S_BRANCH %bb.5
+ bb.5:
+ successors: %bb.5(0x78000000), %bb.6(0x08000000)
+ liveins: $sgpr6
+ %22:vgpr_32 = GLOBAL_LOAD_DWORD %0, 0, 0, implicit $exec :: (load (s32), addrspace 1)
+ %23:vgpr_32 = V_ADD_U32_e32 %22, %6.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %23, 0, 0, implicit $exec :: (store (s32), addrspace 1)
+ %24:vgpr_32 = GLOBAL_LOAD_DWORD %0, 4, 0, implicit $exec :: (load (s32), addrspace 1)
+ %25:vgpr_32 = V_ADD_U32_e32 %24, %6.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %25, 4, 0, implicit $exec :: (store (s32), addrspace 1)
+ %26:vgpr_32 = GLOBAL_LOAD_DWORD %0, 8, 0, implicit $exec :: (load (s32), addrspace 1)
+ %27:vgpr_32 = V_ADD_U32_e32 %26, %6.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %27, 8, 0, implicit $exec :: (store (s32), addrspace 1)
+ %28:vgpr_32 = GLOBAL_LOAD_DWORD %0, 12, 0, implicit $exec :: (load (s32), addrspace 1)
+ %29:vgpr_32 = V_ADD_U32_e32 %28, %6.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %29, 12, 0, implicit $exec :: (store (s32), addrspace 1)
+ %30:vgpr_32 = GLOBAL_LOAD_DWORD %0, 16, 0, implicit $exec :: (load (s32), addrspace 1)
+ %31:vgpr_32 = V_ADD_U32_e32 %30, %11.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %31, 16, 0, implicit $exec :: (store (s32), addrspace 1)
+ %32:vgpr_32 = GLOBAL_LOAD_DWORD %0, 20, 0, implicit $exec :: (load (s32), addrspace 1)
+ %33:vgpr_32 = V_ADD_U32_e32 %32, %11.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %33, 20, 0, implicit $exec :: (store (s32), addrspace 1)
+ %34:vgpr_32 = GLOBAL_LOAD_DWORD %0, 24, 0, implicit $exec :: (load (s32), addrspace 1)
+ %35:vgpr_32 = V_ADD_U32_e32 %34, %11.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %35, 24, 0, implicit $exec :: (store (s32), addrspace 1)
+ %36:vgpr_32 = GLOBAL_LOAD_DWORD %0, 28, 0, implicit $exec :: (load (s32), addrspace 1)
+ %37:vgpr_32 = V_ADD_U32_e32 %36, %11.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %37, 28, 0, implicit $exec :: (store (s32), addrspace 1)
+ %38:vgpr_32 = GLOBAL_LOAD_DWORD %0, 32, 0, implicit $exec :: (load (s32), addrspace 1)
+ %39:vgpr_32 = V_ADD_U32_e32 %38, %16.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %39, 32, 0, implicit $exec :: (store (s32), addrspace 1)
+ %40:vgpr_32 = GLOBAL_LOAD_DWORD %0, 36, 0, implicit $exec :: (load (s32), addrspace 1)
+ %41:vgpr_32 = V_ADD_U32_e32 %40, %16.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %41, 36, 0, implicit $exec :: (store (s32), addrspace 1)
+ %42:vgpr_32 = GLOBAL_LOAD_DWORD %0, 40, 0, implicit $exec :: (load (s32), addrspace 1)
+ %43:vgpr_32 = V_ADD_U32_e32 %42, %16.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %43, 40, 0, implicit $exec :: (store (s32), addrspace 1)
+ %44:vgpr_32 = GLOBAL_LOAD_DWORD %0, 44, 0, implicit $exec :: (load (s32), addrspace 1)
+ %45:vgpr_32 = V_ADD_U32_e32 %44, %16.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %45, 44, 0, implicit $exec :: (store (s32), addrspace 1)
+ %46:vgpr_32 = GLOBAL_LOAD_DWORD %0, 48, 0, implicit $exec :: (load (s32), addrspace 1)
+ %47:vgpr_32 = V_ADD_U32_e32 %46, %21.sub0, implicit $exec
+ GLOBAL_STORE_DWORD %0, %47, 48, 0, implicit $exec :: (store (s32), addrspace 1)
+ %48:vgpr_32 = GLOBAL_LOAD_DWORD %0, 52, 0, implicit $exec :: (load (s32), addrspace 1)
+ %49:vgpr_32 = V_ADD_U32_e32 %48, %21.sub1, implicit $exec
+ GLOBAL_STORE_DWORD %0, %49, 52, 0, implicit $exec :: (store (s32), addrspace 1)
+ %50:vgpr_32 = GLOBAL_LOAD_DWORD %0, 56, 0, implicit $exec :: (load (s32), addrspace 1)
+ %51:vgpr_32 = V_ADD_U32_e32 %50, %21.sub2, implicit $exec
+ GLOBAL_STORE_DWORD %0, %51, 56, 0, implicit $exec :: (store (s32), addrspace 1)
+ %52:vgpr_32 = GLOBAL_LOAD_DWORD %0, 60, 0, implicit $exec :: (load (s32), addrspace 1)
+ %53:vgpr_32 = V_ADD_U32_e32 %52, %21.sub3, implicit $exec
+ GLOBAL_STORE_DWORD %0, %53, 60, 0, implicit $exec :: (store (s32), addrspace 1)
+ $sgpr6 = S_SUB_U32 $sgpr6, 1, implicit-def $scc
+ S_CMP_LG_U32 $sgpr6, 0, implicit-def $scc
+ S_CBRANCH_SCC1 %bb.5, implicit $scc
+ S_BRANCH %bb.6
+ bb.6:
+ S_ENDPGM 0
+...
diff --git a/llvm/unittests/CodeGen/CMakeLists.txt b/llvm/unittests/CodeGen/CMakeLists.txt
index ea8028f4cf979..025a6e657e98d 100644
--- a/llvm/unittests/CodeGen/CMakeLists.txt
+++ b/llvm/unittests/CodeGen/CMakeLists.txt
@@ -33,6 +33,7 @@ add_llvm_unittest(CodeGenTests
InstrRefLDVTest.cpp
LowLevelTypeTest.cpp
LexicalScopesTest.cpp
+ LiveDebugVariablesTest.cpp
MachineBasicBlockTest.cpp
MachineDomTreeUpdaterTest.cpp
MachineInstrBundleIteratorTest.cpp
diff --git a/llvm/unittests/CodeGen/LiveDebugVariablesTest.cpp b/llvm/unittests/CodeGen/LiveDebugVariablesTest.cpp
new file mode 100644
index 0000000000000..6dcee0e177666
--- /dev/null
+++ b/llvm/unittests/CodeGen/LiveDebugVariablesTest.cpp
@@ -0,0 +1,124 @@
+//===- LiveDebugVariablesTest.cpp -----------------------------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "llvm/CodeGen/LiveDebugVariables.h"
+#include "CodeGenTestBase.h"
+#include "llvm/ADT/SmallVector.h"
+#include "llvm/CodeGen/LiveIntervals.h"
+#include "llvm/CodeGen/VirtRegMap.h"
+#include "llvm/Config/Targets.h"
+#include "llvm/Support/TargetSelect.h"
+
+using namespace llvm;
+
+namespace {
+
+class LiveDebugVariablesTest : public CodeGenTestBase {
+public:
+ static void SetUpTestCase() {
+#if LLVM_HAS_X86_TARGET
+ LLVMInitializeX86TargetInfo();
+ LLVMInitializeX86Target();
+ LLVMInitializeX86TargetMC();
+#endif
+ }
+
+ void SetUp() override { setUpImpl("x86_64--", "", ""); }
+};
+
+TEST_F(LiveDebugVariablesTest, PHIsAcrossRepeatedSplitsAndShrink) {
+ ASSERT_TRUE(parseMIR(R"MIR(
+--- |
+ define void @test() !dbg !4 { ret void }
+ !llvm.dbg.cu = !{!0}
+ !llvm.module.flags = !{!5}
+ !0 = distinct !DICompileUnit(language: DW_LANG_C, file: !1, producer: "llvm", isOptimized: true, runtimeVersion: 0, emissionKind: FullDebug)
+ !1 = !DIFile(filename: "test.c", directory: "/")
+ !2 = !DISubroutineType(types: !3)
+ !3 = !{}
+ !4 = distinct !DISubprogram(name: "test", scope: !1, file: !1, line: 1, type: !2, scopeLine: 1, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
+ !5 = !{i32 2, !"Debug Info Version", i32 3}
+...
+---
+name: test
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $eax, $ebx, $ecx
+ %0:gr32 = COPY $eax
+ %1:gr32 = COPY $ebx
+ %2:gr32 = COPY $ecx
+ RET64
+ bb.1:
+ RET64
+ bb.2:
+ RET64
+ bb.3:
+ RET64
+ bb.4:
+ RET64
+...
+)MIR"));
+ MachineFunction &MF = getMF("test");
+ LiveIntervals &LIS = MFAM.getResult<LiveIntervalsAnalysis>(MF);
+ VirtRegMap &VRM = MFAM.getResult<VirtRegMapAnalysis>(MF);
+ Register Old = Register::index2VirtReg(0);
+ Register First = Register::index2VirtReg(1);
+ Register Second = Register::index2VirtReg(2);
+ SmallVector<MCRegister, 3> PhysRegs;
+ for (const MachineInstr &MI : MF.front())
+ if (MI.isCopy())
+ PhysRegs.push_back(MI.getOperand(1).getReg().asMCReg());
+ ASSERT_EQ(PhysRegs.size(), 3u);
+ VRM.assignVirt2Phys(Old, PhysRegs[0]);
+ VRM.assignVirt2Phys(First, PhysRegs[1]);
+ VRM.assignVirt2Phys(Second, PhysRegs[2]);
+
+ // Model five PHIs coalesced into Old, then separated during allocation.
+ LiveInterval &OldLI = LIS.getInterval(Old);
+ OldLI.clear();
+ for (MachineBasicBlock &MBB : MF) {
+ SlotIndex Start = LIS.getMBBStartIdx(&MBB);
+ VNInfo *VNI = OldLI.getNextValue(Start, LIS.getVNInfoAllocator());
+ OldLI.addSegment({Start, LIS.getMBBEndIdx(&MBB), VNI});
+ MF.DebugPHIPositions.try_emplace(MBB.getNumber() + 1, &MBB, Old, 0);
+ }
+ LiveDebugVariables LDV;
+ LDV.analyze(MF, &LIS);
+
+ for (Register Reg : {Old, First, Second})
+ LIS.getInterval(Reg).clear();
+ for (MachineBasicBlock &MBB : MF) {
+ unsigned Block = MBB.getNumber();
+ if (Block == 3)
+ continue;
+ Register Reg = Block == 0 ? First : Block == 1 ? Second : Old;
+ LiveInterval &LI = LIS.getInterval(Reg);
+ SlotIndex Start = LIS.getMBBStartIdx(&MBB);
+ VNInfo *VNI = LI.getNextValue(Start, LIS.getVNInfoAllocator());
+ LI.addSegment({Start, LIS.getMBBEndIdx(&MBB), VNI});
+ }
+
+ LDV.splitRegister(Old, {First}, LIS);
+ LDV.splitRegister(Old, {Second}, LIS);
+ LDV.shrinkRegister(Old);
+ LDV.emitDebugValues(&VRM);
+
+ SmallVector<std::pair<unsigned, Register>, 4> PHIs;
+ for (const MachineBasicBlock &MBB : MF)
+ for (const MachineInstr &MI : MBB)
+ if (MI.isDebugPHI()) {
+ EXPECT_EQ(MI.getOperand(1).getImm(), MBB.getNumber() + 1);
+ PHIs.emplace_back(MI.getOperand(1).getImm(), MI.getOperand(0).getReg());
+ }
+ const SmallVector<std::pair<unsigned, Register>, 4> Expected = {
+ {1, PhysRegs[1]}, {2, PhysRegs[2]}, {3, PhysRegs[0]}, {5, PhysRegs[0]}};
+ EXPECT_EQ(PHIs, Expected);
+}
+
+} // namespace
More information about the llvm-commits
mailing list