[llvm-branch-commits] [llvm] [AMDGPU] Preserve WWM ownership of allocated AGPRs (PR #223278)
Yaxun Liu via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Tue Sep 29 04:24:21 PDT 2026
https://github.com/yxsamliu updated https://github.com/llvm/llvm-project/pull/223278
>From d4e4b148c7cbaa191d6e1c6bb6eca8c7a00825bf Mon Sep 17 00:00:00 2001
From: "Yaxun (Sam) Liu" <yaxun.liu at amd.com>
Date: Tue, 29 Sep 2026 00:19:45 -0400
Subject: [PATCH] [AMDGPU] Preserve WWM ownership of allocated AGPRs
Whole-wave copies preserve every lane of scalar-spill vectors when register
allocation splits their live ranges. A split fragment can be assigned an
AGPR outside the temporary WWM VGPR pool. Keep that register reserved after
the temporary allocation mask is cleared, so later per-lane allocation
cannot reuse it.
Record each WWM copy destination in WWMReservedRegs. Exclude AGPRs from
VGPR compaction, while preserving them through the existing prolog/epilog
spill builders. Distinguish register ownership from spill-slot allocation
in the machine-function state comments.
Test AGPR ownership across the allocation stages, physical AGPR copies on
gfx908 and gfx90a, and the full allocation pipeline for the loop case.
---
llvm/lib/Target/AMDGPU/SIFrameLowering.cpp | 7 +-
llvm/lib/Target/AMDGPU/SILowerWWMCopies.cpp | 11 +--
.../lib/Target/AMDGPU/SIMachineFunctionInfo.h | 16 ++--
.../AMDGPU/preserve-wwm-copy-dst-reg.ll | 10 ++-
.../AMDGPU/si-lower-wwm-agpr-copies.mir | 28 ++++++
.../CodeGen/AMDGPU/si-lower-wwm-copies.mir | 7 +-
.../AMDGPU/whole-wave-register-copy.ll | 16 ++--
.../test/CodeGen/AMDGPU/wwm-agpr-copy-pei.mir | 85 +++++++++++++++++++
.../wwm-regalloc-reserve-split-agpr.mir | 54 ++++++++++++
.../CodeGen/AMDGPU/wwm-spill-undef-loop.mir | 47 ++++++++++
10 files changed, 249 insertions(+), 32 deletions(-)
create mode 100644 llvm/test/CodeGen/AMDGPU/si-lower-wwm-agpr-copies.mir
create mode 100644 llvm/test/CodeGen/AMDGPU/wwm-agpr-copy-pei.mir
create mode 100644 llvm/test/CodeGen/AMDGPU/wwm-regalloc-reserve-split-agpr.mir
create mode 100644 llvm/test/CodeGen/AMDGPU/wwm-spill-undef-loop.mir
diff --git a/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp b/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
index 92fb77b1e6436d..fb68b2881dbb3a 100644
--- a/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFrameLowering.cpp
@@ -1966,11 +1966,10 @@ void SIFrameLowering::determineCalleeSaves(MachineFunction &MF,
SmallVector<Register> SortedWWMVGPRs;
for (Register Reg : MFI->getWWMReservedRegs()) {
- // The shift-back is needed only for the VGPRs used for SGPR spills and they
- // are of 32-bit size. SIPreAllocateWWMRegs pass can add tuples into WWM
- // reserved registers.
+ // The shift-back supports only 32-bit VGPRs. WWM reserved registers may
+ // also contain AGPRs or tuples.
const TargetRegisterClass *RC = TRI->getPhysRegBaseClass(Reg);
- if (TRI->getRegSizeInBits(*RC) != 32)
+ if (!TRI->isVGPRClass(RC) || TRI->getRegSizeInBits(*RC) != 32)
continue;
SortedWWMVGPRs.push_back(Reg);
}
diff --git a/llvm/lib/Target/AMDGPU/SILowerWWMCopies.cpp b/llvm/lib/Target/AMDGPU/SILowerWWMCopies.cpp
index ee3e748b8b36a0..6be435bc3be063 100644
--- a/llvm/lib/Target/AMDGPU/SILowerWWMCopies.cpp
+++ b/llvm/lib/Target/AMDGPU/SILowerWWMCopies.cpp
@@ -38,7 +38,7 @@ class SILowerWWMCopies {
private:
bool isSCCLiveAtMI(const MachineInstr &MI);
- void addToWWMSpills(MachineFunction &MF, Register Reg);
+ void reserveWWMRegister(MachineFunction &MF, Register Reg);
LiveIntervals *LIS;
SlotIndexes *Indexes;
@@ -92,9 +92,9 @@ bool SILowerWWMCopies::isSCCLiveAtMI(const MachineInstr &MI) {
return LR.liveAt(Idx);
}
-// Preserve the inactive lanes of a WWM copy destination at function entry/exit.
-// Fast register allocation has already rewritten virtual registers here.
-void SILowerWWMCopies::addToWWMSpills(MachineFunction &MF, Register Reg) {
+// Record the physical register assigned to a WWM copy destination. It remains
+// owned by the WWM allocation after the temporary allocation mask is cleared.
+void SILowerWWMCopies::reserveWWMRegister(MachineFunction &MF, Register Reg) {
Register PhysReg = Reg;
if (Reg.isVirtual()) {
assert(VRM && "expected VirtRegMap after WWM register allocation");
@@ -102,6 +102,7 @@ void SILowerWWMCopies::addToWWMSpills(MachineFunction &MF, Register Reg) {
assert(PhysReg && "should have allocated a physical register");
}
+ MFI->reserveWWMRegister(PhysReg);
const TargetRegisterClass *RC = TRI->getPhysRegBaseClass(PhysReg);
MFI->allocateWWMSpill(MF, PhysReg, TRI->getSpillSize(*RC),
TRI->getSpillAlign(*RC));
@@ -161,7 +162,7 @@ bool SILowerWWMCopies::run(MachineFunction &MF) {
TII->insertScratchExecCopy(MF, MBB, InsertPt, DL, RegForExecCopy,
isSCCLiveAtMI(MI), Indexes);
TII->restoreExec(MF, MBB, ++InsertPt, DL, RegForExecCopy, Indexes);
- addToWWMSpills(MF, MI.getOperand(0).getReg());
+ reserveWWMRegister(MF, MI.getOperand(0).getReg());
LLVM_DEBUG(dbgs() << "WWM copy manipulation for " << MI);
// Lower WWM_COPY back to COPY
diff --git a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
index 0207c728ea9b80..f9013a36a0c4de 100644
--- a/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
+++ b/llvm/lib/Target/AMDGPU/SIMachineFunctionInfo.h
@@ -573,10 +573,10 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
using WWMSpillsMap = MapVector<Register, int>;
// To track the registers used in instructions that can potentially modify the
// inactive lanes. The WWM instructions and the writelane instructions for
- // spilling SGPRs to VGPRs fall under such category of operations. The VGPRs
- // modified by them should be spilled/restored at function prolog/epilog to
- // avoid any undesired outcome. Each entry in this map holds a pair of values,
- // the VGPR and its stack slot index.
+ // spilling SGPRs to VGPRs fall under such category of operations. The vector
+ // registers modified by them should be spilled/restored at function
+ // prolog/epilog to avoid any undesired outcome. Each entry in this map holds
+ // a pair of values, the register and its stack slot index.
WWMSpillsMap WWMSpills;
// Before allocation, the VGPR registers are partitioned into two distinct
@@ -585,10 +585,10 @@ class SIMachineFunctionInfo final : public AMDGPUMachineFunctionInfo,
BitVector PerLaneVGPRMask;
using ReservedRegSet = SmallSetVector<Register, 8>;
- // To track the VGPRs reserved for WWM instructions. They get stack slots
- // later during PrologEpilogInserter and get added into the superset WWMSpills
- // for actual spilling. A separate set makes the register reserved part and
- // the serialization easier.
+ // To track the vector registers reserved for WWM instructions. WWMSpills
+ // contains the subset that needs prolog/epilog preservation; its spill slots
+ // may be allocated when a register becomes physical or later during PEI. A
+ // separate set makes the register reserved part and serialization easier.
ReservedRegSet WWMReservedRegs;
bool IsWholeWaveFunction = false;
diff --git a/llvm/test/CodeGen/AMDGPU/preserve-wwm-copy-dst-reg.ll b/llvm/test/CodeGen/AMDGPU/preserve-wwm-copy-dst-reg.ll
index c22e48557b8769..1a05ef8c6dd767 100644
--- a/llvm/test/CodeGen/AMDGPU/preserve-wwm-copy-dst-reg.ll
+++ b/llvm/test/CodeGen/AMDGPU/preserve-wwm-copy-dst-reg.ll
@@ -2,10 +2,12 @@
; RUN: llc -mtriple=amdgpu9.06-amd-amdhsa < %s | FileCheck -check-prefix=GFX906 %s
; RUN: llc -mtriple=amdgpu9.08-amd-amdhsa < %s | FileCheck -check-prefix=GFX908 %s
-; Due to high register pressure, regalloc would split the liverange of wwm VGPR register used for SGPR spills
-; and introduce a copy. The copy should be of whole-wave with exec mask manipulation around it.
-; FIXME: The destination register involved in the whole-wave copy should be considered for preserving all the lanes
-; with a spill/restore at function prolog/epilog. The copy might otherwise clobber its inactive lanes unwantedly.
+; Due to high register pressure, regalloc would split the live range of a WWM
+; VGPR used for SGPR spills and introduce a copy. The copy should be whole-wave
+; with EXEC mask manipulation around it.
+; The destination register involved in the whole-wave copy must preserve all
+; lanes with a spill/restore at function prolog/epilog. The copy might otherwise
+; clobber its inactive lanes.
define void @preserve_wwm_copy_dstreg(ptr %parg0, ptr %parg1, ptr %parg2) #0 {
; GFX906-LABEL: preserve_wwm_copy_dstreg:
; GFX906: ; %bb.0:
diff --git a/llvm/test/CodeGen/AMDGPU/si-lower-wwm-agpr-copies.mir b/llvm/test/CodeGen/AMDGPU/si-lower-wwm-agpr-copies.mir
new file mode 100644
index 00000000000000..e44c7fd18c793a
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/si-lower-wwm-agpr-copies.mir
@@ -0,0 +1,28 @@
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx908 -run-pass=liveintervals,virtregmap,si-lower-wwm-copies -verify-machineinstrs -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx908 -passes="require<live-intervals>,require<virtregmap>,si-lower-wwm-copies" -verify-machineinstrs -o - %s | FileCheck %s
+
+---
+name: lower-wwm-agpr-copies
+registers:
+ - { id: 0, class: vgpr_32, flags: [ WWM_REG ] }
+machineFunctionInfo:
+ sgprForEXECCopy: '$sgpr2_sgpr3'
+tracksRegLiveness: true
+body: |
+ bb.0:
+ liveins: $vgpr0
+ ; CHECK-LABEL: name: lower-wwm-agpr-copies
+ ; CHECK: wwmReservedRegs:
+ ; CHECK-NEXT: - '$agpr0'
+ ; CHECK-NEXT: - '$vgpr1'
+ ; CHECK: liveins: $vgpr0
+ ; CHECK-NEXT: {{ $}}
+ ; CHECK-NEXT: $sgpr2_sgpr3 = S_OR_SAVEEXEC_B64 -1, implicit-def $exec, implicit-def dead $scc, implicit $exec
+ ; CHECK-NEXT: $agpr0 = COPY $vgpr0
+ ; CHECK-NEXT: $exec = S_MOV_B64 killed $sgpr2_sgpr3
+ ; CHECK-NEXT: $sgpr2_sgpr3 = S_OR_SAVEEXEC_B64 -1, implicit-def $exec, implicit-def dead $scc, implicit $exec
+ ; CHECK-NEXT: $vgpr1 = COPY $agpr0
+ ; CHECK-NEXT: $exec = S_MOV_B64 killed $sgpr2_sgpr3
+ $agpr0 = WWM_COPY $vgpr0
+ $vgpr1 = WWM_COPY $agpr0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/si-lower-wwm-copies.mir b/llvm/test/CodeGen/AMDGPU/si-lower-wwm-copies.mir
index fc7fe559c1a914..39bbeade4745ce 100644
--- a/llvm/test/CodeGen/AMDGPU/si-lower-wwm-copies.mir
+++ b/llvm/test/CodeGen/AMDGPU/si-lower-wwm-copies.mir
@@ -1,7 +1,7 @@
# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 5
-# RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -run-pass=liveintervals,virtregmap,si-lower-wwm-copies -o - %s | FileCheck %s
-# RUN: llc -mtriple=amdgpu9.00-amd-amdhsa -passes="require<live-intervals>,require<virtregmap>,si-lower-wwm-copies" -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx900 -run-pass=liveintervals,virtregmap,si-lower-wwm-copies -verify-machineinstrs -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx900 -passes="require<live-intervals>,require<virtregmap>,si-lower-wwm-copies" -verify-machineinstrs -o - %s | FileCheck %s
# Check for two cases of $scc being live and dead.
---
@@ -13,6 +13,9 @@ machineFunctionInfo:
tracksRegLiveness: true
body: |
; CHECK-LABEL: name: lower-wwm-copies
+ ; CHECK: wwmReservedRegs:
+ ; CHECK-NEXT: - '$vgpr1'
+ ; CHECK-NEXT: - '$vgpr2'
; CHECK: bb.0:
; CHECK-NEXT: successors: %bb.1(0x80000000)
; CHECK-NEXT: liveins: $vgpr0, $scc
diff --git a/llvm/test/CodeGen/AMDGPU/whole-wave-register-copy.ll b/llvm/test/CodeGen/AMDGPU/whole-wave-register-copy.ll
index bb407023c7440b..6b6374e101fc4a 100644
--- a/llvm/test/CodeGen/AMDGPU/whole-wave-register-copy.ll
+++ b/llvm/test/CodeGen/AMDGPU/whole-wave-register-copy.ll
@@ -1,14 +1,12 @@
; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
-; RUN: llc -mtriple=amdgpu9.0a-amd-amdhsa < %s | FileCheck -check-prefix=GFX90A %s
+; RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -verify-machineinstrs < %s | FileCheck -check-prefix=GFX90A %s
-; The test forces a high vector register pressure and there won't be sufficient VGPRs to be allocated
-; for writelane/readlane SGPR spill instructions. Regalloc would split the vector register liverange
-; by introducing a copy to AGPR register. The VGPR store to AGPR (v_accvgpr_write_b32) and later the
-; restore from AGPR (v_accvgpr_read_b32) should be whole-wave operations and hence exec mask should be
-; manipulated to ensure all lanes are active when these instructions are executed.
-; With the default CSR cost model, an unprofiled single call is too cold to
-; select this split. The profiled loop models repeated execution and makes the
-; stack spills sufficiently expensive.
+; Force high register pressure and whole-wave spilling of scalar-spill vectors.
+; Regalloc splits one WWM live range through an AGPR. The split AGPR remains
+; reserved for WWM after that allocation phase, so later allocation uses a
+; different AGPR.
+; The profiled loop keeps pressure across repeated calls under the default
+; callee-saved register cost model.
define void @vector_reg_liverange_split() #0 {
; GFX90A-LABEL: vector_reg_liverange_split:
; GFX90A: ; %bb.0: ; %entry
diff --git a/llvm/test/CodeGen/AMDGPU/wwm-agpr-copy-pei.mir b/llvm/test/CodeGen/AMDGPU/wwm-agpr-copy-pei.mir
new file mode 100644
index 00000000000000..8fbcb5cf83161e
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/wwm-agpr-copy-pei.mir
@@ -0,0 +1,85 @@
+# NOTE: Assertions have been autogenerated by utils/update_mir_test_checks.py UTC_ARGS: --version 6
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -run-pass=liveintervals,virtregmap,si-lower-wwm-copies,virtregrewriter,prolog-epilog -verify-machineinstrs -o - %s | FileCheck %s --check-prefix=GFX90A
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx908 -run-pass=liveintervals,virtregmap,si-lower-wwm-copies,virtregrewriter,prolog-epilog -verify-machineinstrs -o - %s | FileCheck %s --check-prefix=GFX908
+
+# Preserve the AGPR and compact the VGPR together with its existing spill slot.
+
+---
+name: physical_wwm_agpr_copy_pei
+tracksRegLiveness: true
+noVRegs: false
+isSSA: false
+registers:
+ # Keep VRegFlags non-empty so si-lower-wwm-copies processes the physical
+ # WWM_COPY instructions below, as it does after WWM register allocation.
+ - { id: 0, class: vgpr_32, flags: [ WWM_REG ] }
+frameInfo:
+ maxAlignment: 4
+machineFunctionInfo:
+ isEntryFunction: false
+ scratchRSrcReg: '$sgpr0_sgpr1_sgpr2_sgpr3'
+ stackPtrOffsetReg: '$sgpr32'
+ frameOffsetReg: '$sgpr33'
+ sgprForEXECCopy: '$sgpr4_sgpr5'
+body: |
+ bb.0:
+ liveins: $vgpr63, $sgpr30_sgpr31
+ ; GFX90A-LABEL: name: physical_wwm_agpr_copy_pei
+ ; GFX90A: liveins: $vgpr63, $sgpr30_sgpr31, $agpr0
+ ; GFX90A-NEXT: {{ $}}
+ ; GFX90A-NEXT: frame-setup CFI_INSTRUCTION llvm_def_aspace_cfa $sgpr32, 0, 6
+ ; GFX90A-NEXT: frame-setup CFI_INSTRUCTION llvm_register_pair $pc_reg, $sgpr30, 32, $sgpr31, 32
+ ; GFX90A-NEXT: frame-setup CFI_INSTRUCTION undefined $vgpr0
+ ; GFX90A-NEXT: frame-setup CFI_INSTRUCTION undefined $agpr0
+ ; GFX90A-NEXT: frame-setup CFI_INSTRUCTION undefined $sgpr6
+ ; GFX90A-NEXT: frame-setup CFI_INSTRUCTION undefined $sgpr7
+ ; GFX90A-NEXT: $sgpr6_sgpr7 = S_XOR_SAVEEXEC_B64 -1, implicit-def $exec, implicit-def dead $scc, implicit $exec
+ ; GFX90A-NEXT: BUFFER_STORE_DWORD_OFFSET $agpr0, $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr32, 0, 0, 0, implicit $exec :: ("amdgpu-thread-private" store (s32) into %stack.0, addrspace 5)
+ ; GFX90A-NEXT: frame-setup CFI_INSTRUCTION offset $agpr0, 0
+ ; GFX90A-NEXT: BUFFER_STORE_DWORD_OFFSET killed $vgpr0, $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr32, 4, 0, 0, implicit $exec :: ("amdgpu-thread-private" store (s32) into %stack.1, addrspace 5)
+ ; GFX90A-NEXT: frame-setup CFI_INSTRUCTION offset $vgpr0, 256
+ ; GFX90A-NEXT: $exec = S_MOV_B64 killed $sgpr6_sgpr7
+ ; GFX90A-NEXT: $sgpr6_sgpr7 = S_OR_SAVEEXEC_B64 -1, implicit-def $exec, implicit-def dead $scc, implicit $exec
+ ; GFX90A-NEXT: $agpr0 = COPY $vgpr63
+ ; GFX90A-NEXT: $exec = S_MOV_B64 killed $sgpr6_sgpr7
+ ; GFX90A-NEXT: $sgpr6_sgpr7 = S_OR_SAVEEXEC_B64 -1, implicit-def $exec, implicit-def dead $scc, implicit $exec
+ ; GFX90A-NEXT: $vgpr0 = COPY $agpr0
+ ; GFX90A-NEXT: $exec = S_MOV_B64 killed $sgpr6_sgpr7
+ ; GFX90A-NEXT: $sgpr6_sgpr7 = S_XOR_SAVEEXEC_B64 -1, implicit-def $exec, implicit-def dead $scc, implicit $exec
+ ; GFX90A-NEXT: $agpr0 = BUFFER_LOAD_DWORD_OFFSET $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr32, 0, 0, 0, implicit $exec :: ("amdgpu-thread-private" load (s32) from %stack.0, addrspace 5)
+ ; GFX90A-NEXT: $vgpr0 = BUFFER_LOAD_DWORD_OFFSET $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr32, 4, 0, 0, implicit $exec :: ("amdgpu-thread-private" load (s32) from %stack.1, addrspace 5)
+ ; GFX90A-NEXT: $exec = S_MOV_B64 killed $sgpr6_sgpr7
+ ; GFX90A-NEXT: S_SETPC_B64_return $sgpr30_sgpr31
+ ;
+ ; GFX908-LABEL: name: physical_wwm_agpr_copy_pei
+ ; GFX908: liveins: $vgpr63, $sgpr30_sgpr31, $agpr0
+ ; GFX908-NEXT: {{ $}}
+ ; GFX908-NEXT: frame-setup CFI_INSTRUCTION llvm_def_aspace_cfa $sgpr32, 0, 6
+ ; GFX908-NEXT: frame-setup CFI_INSTRUCTION llvm_register_pair $pc_reg, $sgpr30, 32, $sgpr31, 32
+ ; GFX908-NEXT: frame-setup CFI_INSTRUCTION undefined $vgpr0
+ ; GFX908-NEXT: frame-setup CFI_INSTRUCTION undefined $agpr0
+ ; GFX908-NEXT: frame-setup CFI_INSTRUCTION undefined $sgpr6
+ ; GFX908-NEXT: frame-setup CFI_INSTRUCTION undefined $sgpr7
+ ; GFX908-NEXT: $sgpr6_sgpr7 = S_XOR_SAVEEXEC_B64 -1, implicit-def $exec, implicit-def dead $scc, implicit $exec
+ ; GFX908-NEXT: $vgpr63 = V_ACCVGPR_READ_B32_e64 $agpr0, implicit $exec
+ ; GFX908-NEXT: BUFFER_STORE_DWORD_OFFSET $vgpr63, $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr32, 0, 0, 0, implicit $exec :: ("amdgpu-thread-private" store (s32) into %stack.0, addrspace 5)
+ ; GFX908-NEXT: frame-setup CFI_INSTRUCTION offset $agpr0, 0
+ ; GFX908-NEXT: BUFFER_STORE_DWORD_OFFSET killed $vgpr0, $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr32, 4, 0, 0, implicit $exec :: ("amdgpu-thread-private" store (s32) into %stack.1, addrspace 5)
+ ; GFX908-NEXT: frame-setup CFI_INSTRUCTION offset $vgpr0, 256
+ ; GFX908-NEXT: $exec = S_MOV_B64 killed $sgpr6_sgpr7
+ ; GFX908-NEXT: $sgpr6_sgpr7 = S_OR_SAVEEXEC_B64 -1, implicit-def $exec, implicit-def dead $scc, implicit $exec
+ ; GFX908-NEXT: $agpr0 = COPY $vgpr63
+ ; GFX908-NEXT: $exec = S_MOV_B64 killed $sgpr6_sgpr7
+ ; GFX908-NEXT: $sgpr6_sgpr7 = S_OR_SAVEEXEC_B64 -1, implicit-def $exec, implicit-def dead $scc, implicit $exec
+ ; GFX908-NEXT: $vgpr0 = COPY $agpr0
+ ; GFX908-NEXT: $exec = S_MOV_B64 killed $sgpr6_sgpr7
+ ; GFX908-NEXT: $sgpr6_sgpr7 = S_XOR_SAVEEXEC_B64 -1, implicit-def $exec, implicit-def dead $scc, implicit $exec
+ ; GFX908-NEXT: $vgpr63 = BUFFER_LOAD_DWORD_OFFSET $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr32, 0, 0, 0, implicit $exec :: ("amdgpu-thread-private" load (s32) from %stack.0, addrspace 5)
+ ; GFX908-NEXT: $agpr0 = V_ACCVGPR_WRITE_B32_e64 killed $vgpr63, implicit $exec
+ ; GFX908-NEXT: $vgpr0 = BUFFER_LOAD_DWORD_OFFSET $sgpr0_sgpr1_sgpr2_sgpr3, $sgpr32, 4, 0, 0, implicit $exec :: ("amdgpu-thread-private" load (s32) from %stack.1, addrspace 5)
+ ; GFX908-NEXT: $exec = S_MOV_B64 killed $sgpr6_sgpr7
+ ; GFX908-NEXT: S_SETPC_B64_return $sgpr30_sgpr31
+ $agpr0 = WWM_COPY $vgpr63
+ $vgpr62 = WWM_COPY $agpr0
+ S_SETPC_B64_return $sgpr30_sgpr31
+...
diff --git a/llvm/test/CodeGen/AMDGPU/wwm-regalloc-reserve-split-agpr.mir b/llvm/test/CodeGen/AMDGPU/wwm-regalloc-reserve-split-agpr.mir
new file mode 100644
index 00000000000000..1c25ff54f4bdb0
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/wwm-regalloc-reserve-split-agpr.mir
@@ -0,0 +1,54 @@
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx908 -start-before=si-lower-sgpr-spills -stop-after=virtregrewriter,2 -amdgpu-num-vgprs-for-wwm-alloc=1 -verify-machineinstrs -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx90a -start-before=si-lower-sgpr-spills -stop-after=virtregrewriter,2 -amdgpu-num-vgprs-for-wwm-alloc=1 -verify-machineinstrs -o - %s | FileCheck %s
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx950 -start-before=si-lower-sgpr-spills -stop-after=virtregrewriter,2 -amdgpu-num-vgprs-for-wwm-alloc=1 -verify-machineinstrs -o - %s | FileCheck %s
+
+# With two overlapping scalar-spill vectors and one WWM VGPR, splitting may
+# place a WWM fragment in an AGPR. That AGPR must remain reserved when the
+# per-lane allocator runs, so the later explicit AGPR value uses a different
+# register.
+# CHECK-LABEL: name: wwm_pool_reserves_split_agpr
+# CHECK: wwmReservedRegs:
+# CHECK-DAG: - '$agpr0'
+# CHECK-DAG: - '$vgpr{{[0-9]+}}'
+# CHECK: $agpr0 = lr-split COPY
+# CHECK: renamable $agpr1 = V_ACCVGPR_WRITE_B32_e64
+# CHECK-NEXT: INLINEASM &"; use $0", sideeffect attdialect, reguse:AGPR_32, killed renamable $agpr1
+# CHECK-NEXT: S_ENDPGM
+
+--- |
+ define amdgpu_kernel void @wwm_pool_reserves_split_agpr() #0 {
+ ret void
+ }
+
+ attributes #0 = { "amdgpu-no-workitem-id-x" "amdgpu-no-workitem-id-y" "amdgpu-no-workitem-id-z" }
+...
+---
+name: wwm_pool_reserves_split_agpr
+tracksRegLiveness: true
+frameInfo:
+ maxAlignment: 4
+stack:
+ - { id: 0, type: spill-slot, size: 128, alignment: 4, stack-id: sgpr-spill }
+ - { id: 1, type: spill-slot, size: 128, alignment: 4, stack-id: sgpr-spill }
+ - { id: 2, type: spill-slot, size: 4, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ isEntryFunction: true
+ scratchRSrcReg: '$sgpr0_sgpr1_sgpr2_sgpr3'
+ stackPtrOffsetReg: '$sgpr32'
+ frameOffsetReg: '$sgpr33'
+ hasSpilledSGPRs: true
+body: |
+ bb.0:
+ liveins: $sgpr34, $sgpr36_sgpr37_sgpr38_sgpr39_sgpr40_sgpr41_sgpr42_sgpr43_sgpr44_sgpr45_sgpr46_sgpr47_sgpr48_sgpr49_sgpr50_sgpr51_sgpr52_sgpr53_sgpr54_sgpr55_sgpr56_sgpr57_sgpr58_sgpr59_sgpr60_sgpr61_sgpr62_sgpr63_sgpr64_sgpr65_sgpr66_sgpr67, $sgpr68_sgpr69_sgpr70_sgpr71_sgpr72_sgpr73_sgpr74_sgpr75_sgpr76_sgpr77_sgpr78_sgpr79_sgpr80_sgpr81_sgpr82_sgpr83_sgpr84_sgpr85_sgpr86_sgpr87_sgpr88_sgpr89_sgpr90_sgpr91_sgpr92_sgpr93_sgpr94_sgpr95_sgpr96_sgpr97_sgpr98_sgpr99
+
+ SI_SPILL_S1024_SAVE killed $sgpr36_sgpr37_sgpr38_sgpr39_sgpr40_sgpr41_sgpr42_sgpr43_sgpr44_sgpr45_sgpr46_sgpr47_sgpr48_sgpr49_sgpr50_sgpr51_sgpr52_sgpr53_sgpr54_sgpr55_sgpr56_sgpr57_sgpr58_sgpr59_sgpr60_sgpr61_sgpr62_sgpr63_sgpr64_sgpr65_sgpr66_sgpr67, %stack.0, implicit $exec, implicit $sgpr0_sgpr1_sgpr2_sgpr3, implicit $sgpr32
+ SI_SPILL_S1024_SAVE killed $sgpr68_sgpr69_sgpr70_sgpr71_sgpr72_sgpr73_sgpr74_sgpr75_sgpr76_sgpr77_sgpr78_sgpr79_sgpr80_sgpr81_sgpr82_sgpr83_sgpr84_sgpr85_sgpr86_sgpr87_sgpr88_sgpr89_sgpr90_sgpr91_sgpr92_sgpr93_sgpr94_sgpr95_sgpr96_sgpr97_sgpr98_sgpr99, %stack.1, implicit $exec, implicit $sgpr0_sgpr1_sgpr2_sgpr3, implicit $sgpr32
+ SI_SPILL_S32_SAVE killed $sgpr34, %stack.2, implicit $exec, implicit $sgpr0_sgpr1_sgpr2_sgpr3, implicit $sgpr32
+ renamable $sgpr34 = SI_SPILL_S32_RESTORE %stack.2, implicit $exec, implicit $sgpr0_sgpr1_sgpr2_sgpr3, implicit $sgpr32
+ renamable $sgpr68_sgpr69_sgpr70_sgpr71_sgpr72_sgpr73_sgpr74_sgpr75_sgpr76_sgpr77_sgpr78_sgpr79_sgpr80_sgpr81_sgpr82_sgpr83_sgpr84_sgpr85_sgpr86_sgpr87_sgpr88_sgpr89_sgpr90_sgpr91_sgpr92_sgpr93_sgpr94_sgpr95_sgpr96_sgpr97_sgpr98_sgpr99 = SI_SPILL_S1024_RESTORE %stack.1, implicit $exec, implicit $sgpr0_sgpr1_sgpr2_sgpr3, implicit $sgpr32
+ renamable $sgpr36_sgpr37_sgpr38_sgpr39_sgpr40_sgpr41_sgpr42_sgpr43_sgpr44_sgpr45_sgpr46_sgpr47_sgpr48_sgpr49_sgpr50_sgpr51_sgpr52_sgpr53_sgpr54_sgpr55_sgpr56_sgpr57_sgpr58_sgpr59_sgpr60_sgpr61_sgpr62_sgpr63_sgpr64_sgpr65_sgpr66_sgpr67 = SI_SPILL_S1024_RESTORE %stack.0, implicit $exec, implicit $sgpr0_sgpr1_sgpr2_sgpr3, implicit $sgpr32
+ %0:vgpr_32 = V_MOV_B32_e32 42, implicit $exec
+ %1:agpr_32 = V_ACCVGPR_WRITE_B32_e64 %0, implicit $exec
+ INLINEASM &"; use $0", sideeffect attdialect, reguse:AGPR_32, %1
+ S_ENDPGM 0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/wwm-spill-undef-loop.mir b/llvm/test/CodeGen/AMDGPU/wwm-spill-undef-loop.mir
new file mode 100644
index 00000000000000..b1a841ec9a1fe0
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/wwm-spill-undef-loop.mir
@@ -0,0 +1,47 @@
+# RUN: llc -mtriple=amdgcn-amd-amdhsa -mcpu=gfx908 -start-before=si-lower-sgpr-spills -verify-machineinstrs -filetype=null %s
+
+# Exercise the complete allocation pipeline, including EXEC-save setup.
+
+---
+name: widget
+tracksRegLiveness: true
+frameInfo:
+ adjustsStack: true
+stack:
+ - { id: 0, type: spill-slot, size: 4, alignment: 4, stack-id: sgpr-spill }
+ - { id: 1, type: spill-slot, size: 4, alignment: 4, stack-id: sgpr-spill }
+machineFunctionInfo:
+ hasSpilledSGPRs: true
+ scratchRSrcReg: '$sgpr0_sgpr1_sgpr2_sgpr3'
+ stackPtrOffsetReg: '$sgpr32'
+body: |
+ bb.0:
+ liveins: $sgpr12, $sgpr13, $sgpr14, $sgpr15
+
+ %0:vgpr_32 = IMPLICIT_DEF
+ SI_SPILL_S32_SAVE $sgpr15, %stack.0, implicit $exec, implicit $sgpr32 :: (store (s32) into %stack.0, addrspace 5)
+ %1:vgpr_32 = V_AND_B32_e32 1, %0, implicit $exec
+
+ bb.1:
+ successors: %bb.3, %bb.2
+
+ S_CBRANCH_EXECZ %bb.2, implicit $exec
+ S_BRANCH %bb.3
+
+ bb.2:
+ successors: %bb.4(0x04000000), %bb.1(0x7c000000)
+ liveins: $sgpr86, $sgpr87, $sgpr66_sgpr67, $sgpr68_sgpr69, $sgpr70_sgpr71, $sgpr80_sgpr81, $sgpr82_sgpr83, $sgpr84_sgpr85, $sgpr96_sgpr97, $sgpr98_sgpr99
+
+ S_CBRANCH_EXECNZ %bb.1, implicit $exec
+ S_BRANCH %bb.4
+
+ bb.3:
+ ADJCALLSTACKUP 0, 0, implicit-def dead $scc, implicit-def $sgpr32, implicit $sgpr32
+ $sgpr14 = SI_SPILL_S32_RESTORE %stack.1, implicit $exec, implicit $sgpr32 :: (load (s32) from %stack.1, addrspace 5)
+ ADJCALLSTACKDOWN 0, 28, implicit-def dead $scc, implicit-def $sgpr32, implicit $sgpr32
+ S_BRANCH %bb.2
+
+ bb.4:
+ SI_RETURN
+
+...
More information about the llvm-branch-commits
mailing list