[llvm] [AMDGPU] Don't merge M0 initializations across calls that clobber M0 (PR #221963)
Pankaj Dwivedi via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 8 04:26:42 PDT 2026
https://github.com/PankajDwivedi-25 created https://github.com/llvm/llvm-project/pull/221963
hoistAndMergeSGPRInits collected clobbers only from MRI.def_instructions(M0),
which lists explicit defs. Calls clobber M0 via a regmask and were therefore
invisible, so M0 inits were merged and hoisted across them and the required
re-init after a call was dropped. Scan the function using modifiesRegister.
>From 7d98df884807f34aab2461703b9d71cdd4e2dd46 Mon Sep 17 00:00:00 2001
From: padivedi <pankajkumar.divedi at amd.com>
Date: Tue, 8 Sep 2026 16:53:01 +0530
Subject: [PATCH] [AMDGPU] Don't merge M0 initializations across calls that
clobber M0
---
llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp | 23 +++-
llvm/test/CodeGen/AMDGPU/ds_read2.ll | 1 +
llvm/test/CodeGen/AMDGPU/merge-m0.mir | 133 +++++++++++++++++++++
3 files changed, 152 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
index b0c896b1a122a..7028f89a49f1c 100644
--- a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
@@ -457,7 +457,7 @@ getFirstNonPrologue(MachineBasicBlock *MBB, const TargetInstrInfo *TII) {
// This is intended to combine M0 initializations, but can work with any
// SGPR. A VGPR cannot be processed since we cannot guarantee vector
// executioon.
-static bool hoistAndMergeSGPRInits(unsigned Reg,
+static bool hoistAndMergeSGPRInits(unsigned Reg, MachineFunction &MF,
const MachineRegisterInfo &MRI,
const TargetRegisterInfo *TRI,
MachineDominatorTree &MDT,
@@ -469,12 +469,15 @@ static bool hoistAndMergeSGPRInits(unsigned Reg,
SmallVector<MachineInstr*, 8> Clobbers;
// List of instructions marked for deletion.
SmallPtrSet<MachineInstr *, 8> MergedInstrs;
+ // Instructions already classified as an init or a clobber.
+ SmallPtrSet<MachineInstr *, 16> Classified;
bool Changed = false;
- for (auto &MI : MRI.def_instructions(Reg)) {
+ for (MachineInstr &MI : MRI.def_instructions(Reg)) {
+ Classified.insert(&MI);
MachineOperand *Imm = nullptr;
- for (auto &MO : MI.operands()) {
+ for (MachineOperand &MO : MI.operands()) {
if ((MO.isReg() && ((MO.isDef() && MO.getReg() != Reg) || !MO.isDef())) ||
(!MO.isImm() && !MO.isReg()) || (MO.isImm() && Imm)) {
Imm = nullptr;
@@ -489,6 +492,16 @@ static bool hoistAndMergeSGPRInits(unsigned Reg,
Clobbers.push_back(&MI);
}
+ if (Inits.empty())
+ return false;
+
+ // Calls clobber Reg with a regmask operand instead of an explicit def, so
+ // they do not appear on the def list of Reg.
+ for (MachineBasicBlock &MBB : MF)
+ for (MachineInstr &MI : MBB)
+ if (!Classified.contains(&MI) && MI.modifiesRegister(Reg, TRI))
+ Clobbers.push_back(&MI);
+
for (auto &Init : Inits) {
auto &Defs = Init.second;
@@ -609,7 +622,7 @@ static bool hoistAndMergeSGPRInits(unsigned Reg,
const unsigned Threshold = 50;
// Search until B or Threshold for a place to insert the initialization.
for (unsigned I = 0; R != B && I < Threshold; ++R, ++I)
- if (R->readsRegister(Reg, TRI) || R->definesRegister(Reg, TRI) ||
+ if (R->readsRegister(Reg, TRI) || R->modifiesRegister(Reg, TRI) ||
TII->isSchedulingBoundary(*R, MBB, *MBB->getParent()))
break;
@@ -811,7 +824,7 @@ bool SIFixSGPRCopies::run(MachineFunction &MF) {
TII->legalizeOperands(*Relegalize.pop_back_val(), MDT);
if (MF.getTarget().getOptLevel() > CodeGenOptLevel::None && EnableM0Merge)
- hoistAndMergeSGPRInits(AMDGPU::M0, *MRI, TRI, *MDT, TII);
+ hoistAndMergeSGPRInits(AMDGPU::M0, MF, *MRI, TRI, *MDT, TII);
SiblingPenalty.clear();
V2SCopies.clear();
diff --git a/llvm/test/CodeGen/AMDGPU/ds_read2.ll b/llvm/test/CodeGen/AMDGPU/ds_read2.ll
index 16b451699a9b1..1c20591ecebfe 100644
--- a/llvm/test/CodeGen/AMDGPU/ds_read2.ll
+++ b/llvm/test/CodeGen/AMDGPU/ds_read2.ll
@@ -1360,6 +1360,7 @@ define amdgpu_kernel void @ds_read_call_read(ptr addrspace(1) %out, ptr addrspac
; CI-NEXT: s_mov_b32 s39, 0xf000
; CI-NEXT: s_mov_b32 s38, -1
; CI-NEXT: s_swappc_b64 s[30:31], s[16:17]
+; CI-NEXT: s_mov_b32 m0, -1
; CI-NEXT: ds_read_b32 v0, v40 offset:4
; CI-NEXT: s_waitcnt lgkmcnt(0)
; CI-NEXT: v_add_i32_e32 v0, vcc, v41, v0
diff --git a/llvm/test/CodeGen/AMDGPU/merge-m0.mir b/llvm/test/CodeGen/AMDGPU/merge-m0.mir
index a06ffafa7a046..2f6708028158c 100644
--- a/llvm/test/CodeGen/AMDGPU/merge-m0.mir
+++ b/llvm/test/CodeGen/AMDGPU/merge-m0.mir
@@ -678,3 +678,136 @@ body: |
bb.3:
S_ENDPGM 0
...
+
+# A call clobbers m0 with a regmask instead of an explicit def, so the init
+# after the call must be kept.
+
+# GCN-LABEL: name: merge-m0-call-clobber
+# GCN: SI_INIT_M0 -1
+# GCN: DS_WRITE_B32
+# GCN-NEXT: SI_CALL
+# GCN-NEXT: SI_INIT_M0 -1
+# GCN-NEXT: DS_WRITE_B32
+# GCN-NEXT: S_ENDPGM
+
+---
+name: merge-m0-call-clobber
+registers:
+ - { id: 0, class: vgpr_32 }
+ - { id: 1, class: vgpr_32 }
+ - { id: 2, class: sreg_64_xexec }
+body: |
+ bb.0:
+ %0 = IMPLICIT_DEF
+ %1 = IMPLICIT_DEF
+ %2 = IMPLICIT_DEF
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ dead $sgpr30_sgpr31 = SI_CALL %2, 0, csr_amdgpu
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ S_ENDPGM 0
+...
+
+# The inits cannot be hoisted into the common dominator because the call in
+# that block clobbers m0.
+
+# GCN-LABEL: name: merge-m0-call-clobber-cross-block
+# GCN: bb.0:
+# GCN-NOT: SI_INIT_M0
+# GCN: SI_CALL
+# GCN: bb.1:
+# GCN: SI_INIT_M0 -1
+# GCN-NEXT: DS_WRITE_B32
+# GCN: bb.2:
+# GCN: SI_INIT_M0 -1
+# GCN-NEXT: DS_WRITE_B32
+
+---
+name: merge-m0-call-clobber-cross-block
+registers:
+ - { id: 0, class: vgpr_32 }
+ - { id: 1, class: vgpr_32 }
+ - { id: 2, class: sreg_64_xexec }
+body: |
+ bb.0:
+ successors: %bb.1, %bb.2
+
+ %0 = IMPLICIT_DEF
+ %1 = IMPLICIT_DEF
+ %2 = IMPLICIT_DEF
+ dead $sgpr30_sgpr31 = SI_CALL %2, 0, csr_amdgpu
+ S_CBRANCH_VCCZ %bb.1, implicit undef $vcc
+ S_BRANCH %bb.2
+
+ bb.1:
+ successors: %bb.3
+
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ S_BRANCH %bb.3
+
+ bb.2:
+ successors: %bb.3
+
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ S_BRANCH %bb.3
+
+ bb.3:
+ S_ENDPGM 0
+...
+
+# Both inits are dominated by the call, so they still merge, but the surviving
+# init must not move above the call.
+
+# GCN-LABEL: name: merge-m0-call-dominates-inits
+# GCN: SI_CALL
+# GCN-NEXT: SI_INIT_M0 -1
+# GCN-NEXT: DS_WRITE_B32
+# GCN-NEXT: DS_WRITE_B32
+# GCN-NEXT: S_ENDPGM
+
+---
+name: merge-m0-call-dominates-inits
+registers:
+ - { id: 0, class: vgpr_32 }
+ - { id: 1, class: vgpr_32 }
+ - { id: 2, class: sreg_64_xexec }
+body: |
+ bb.0:
+ %0 = IMPLICIT_DEF
+ %1 = IMPLICIT_DEF
+ %2 = IMPLICIT_DEF
+ dead $sgpr30_sgpr31 = SI_CALL %2, 0, csr_amdgpu
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ S_ENDPGM 0
+...
+
+# A single init must not be scheduled above a call.
+
+# GCN-LABEL: name: move-m0-call-clobber
+# GCN: SI_CALL
+# GCN-NEXT: SI_INIT_M0 -1
+# GCN-NEXT: DS_WRITE_B32
+# GCN-NEXT: S_ENDPGM
+
+---
+name: move-m0-call-clobber
+registers:
+ - { id: 0, class: vgpr_32 }
+ - { id: 1, class: vgpr_32 }
+ - { id: 2, class: sreg_64_xexec }
+body: |
+ bb.0:
+ %0 = IMPLICIT_DEF
+ %1 = IMPLICIT_DEF
+ %2 = IMPLICIT_DEF
+ dead $sgpr30_sgpr31 = SI_CALL %2, 0, csr_amdgpu
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ S_ENDPGM 0
+...
More information about the llvm-commits
mailing list