[llvm] [AMDGPU] Don't merge M0 initializations across calls that clobber M0 (PR #221963)
Pankaj Dwivedi via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 10 05:00:48 PDT 2026
https://github.com/PankajDwivedi-25 updated https://github.com/llvm/llvm-project/pull/221963
>From 19a9fd6cdc569fa20f8a43a257bf45b6981a77cb Mon Sep 17 00:00:00 2001
From: padivedi <pankajkumar.divedi at amd.com>
Date: Tue, 8 Sep 2026 16:53:01 +0530
Subject: [PATCH 1/2] [AMDGPU] Don't merge M0 initializations across calls that
clobber M0
---
llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp | 23 +++-
llvm/test/CodeGen/AMDGPU/ds_read2.ll | 1 +
llvm/test/CodeGen/AMDGPU/merge-m0.mir | 133 +++++++++++++++++++++
3 files changed, 152 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
index b0c896b1a122a..7028f89a49f1c 100644
--- a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
@@ -457,7 +457,7 @@ getFirstNonPrologue(MachineBasicBlock *MBB, const TargetInstrInfo *TII) {
// This is intended to combine M0 initializations, but can work with any
// SGPR. A VGPR cannot be processed since we cannot guarantee vector
// executioon.
-static bool hoistAndMergeSGPRInits(unsigned Reg,
+static bool hoistAndMergeSGPRInits(unsigned Reg, MachineFunction &MF,
const MachineRegisterInfo &MRI,
const TargetRegisterInfo *TRI,
MachineDominatorTree &MDT,
@@ -469,12 +469,15 @@ static bool hoistAndMergeSGPRInits(unsigned Reg,
SmallVector<MachineInstr*, 8> Clobbers;
// List of instructions marked for deletion.
SmallPtrSet<MachineInstr *, 8> MergedInstrs;
+ // Instructions already classified as an init or a clobber.
+ SmallPtrSet<MachineInstr *, 16> Classified;
bool Changed = false;
- for (auto &MI : MRI.def_instructions(Reg)) {
+ for (MachineInstr &MI : MRI.def_instructions(Reg)) {
+ Classified.insert(&MI);
MachineOperand *Imm = nullptr;
- for (auto &MO : MI.operands()) {
+ for (MachineOperand &MO : MI.operands()) {
if ((MO.isReg() && ((MO.isDef() && MO.getReg() != Reg) || !MO.isDef())) ||
(!MO.isImm() && !MO.isReg()) || (MO.isImm() && Imm)) {
Imm = nullptr;
@@ -489,6 +492,16 @@ static bool hoistAndMergeSGPRInits(unsigned Reg,
Clobbers.push_back(&MI);
}
+ if (Inits.empty())
+ return false;
+
+ // Calls clobber Reg with a regmask operand instead of an explicit def, so
+ // they do not appear on the def list of Reg.
+ for (MachineBasicBlock &MBB : MF)
+ for (MachineInstr &MI : MBB)
+ if (!Classified.contains(&MI) && MI.modifiesRegister(Reg, TRI))
+ Clobbers.push_back(&MI);
+
for (auto &Init : Inits) {
auto &Defs = Init.second;
@@ -609,7 +622,7 @@ static bool hoistAndMergeSGPRInits(unsigned Reg,
const unsigned Threshold = 50;
// Search until B or Threshold for a place to insert the initialization.
for (unsigned I = 0; R != B && I < Threshold; ++R, ++I)
- if (R->readsRegister(Reg, TRI) || R->definesRegister(Reg, TRI) ||
+ if (R->readsRegister(Reg, TRI) || R->modifiesRegister(Reg, TRI) ||
TII->isSchedulingBoundary(*R, MBB, *MBB->getParent()))
break;
@@ -811,7 +824,7 @@ bool SIFixSGPRCopies::run(MachineFunction &MF) {
TII->legalizeOperands(*Relegalize.pop_back_val(), MDT);
if (MF.getTarget().getOptLevel() > CodeGenOptLevel::None && EnableM0Merge)
- hoistAndMergeSGPRInits(AMDGPU::M0, *MRI, TRI, *MDT, TII);
+ hoistAndMergeSGPRInits(AMDGPU::M0, MF, *MRI, TRI, *MDT, TII);
SiblingPenalty.clear();
V2SCopies.clear();
diff --git a/llvm/test/CodeGen/AMDGPU/ds_read2.ll b/llvm/test/CodeGen/AMDGPU/ds_read2.ll
index 16b451699a9b1..1c20591ecebfe 100644
--- a/llvm/test/CodeGen/AMDGPU/ds_read2.ll
+++ b/llvm/test/CodeGen/AMDGPU/ds_read2.ll
@@ -1360,6 +1360,7 @@ define amdgpu_kernel void @ds_read_call_read(ptr addrspace(1) %out, ptr addrspac
; CI-NEXT: s_mov_b32 s39, 0xf000
; CI-NEXT: s_mov_b32 s38, -1
; CI-NEXT: s_swappc_b64 s[30:31], s[16:17]
+; CI-NEXT: s_mov_b32 m0, -1
; CI-NEXT: ds_read_b32 v0, v40 offset:4
; CI-NEXT: s_waitcnt lgkmcnt(0)
; CI-NEXT: v_add_i32_e32 v0, vcc, v41, v0
diff --git a/llvm/test/CodeGen/AMDGPU/merge-m0.mir b/llvm/test/CodeGen/AMDGPU/merge-m0.mir
index a06ffafa7a046..2f6708028158c 100644
--- a/llvm/test/CodeGen/AMDGPU/merge-m0.mir
+++ b/llvm/test/CodeGen/AMDGPU/merge-m0.mir
@@ -678,3 +678,136 @@ body: |
bb.3:
S_ENDPGM 0
...
+
+# A call clobbers m0 with a regmask instead of an explicit def, so the init
+# after the call must be kept.
+
+# GCN-LABEL: name: merge-m0-call-clobber
+# GCN: SI_INIT_M0 -1
+# GCN: DS_WRITE_B32
+# GCN-NEXT: SI_CALL
+# GCN-NEXT: SI_INIT_M0 -1
+# GCN-NEXT: DS_WRITE_B32
+# GCN-NEXT: S_ENDPGM
+
+---
+name: merge-m0-call-clobber
+registers:
+ - { id: 0, class: vgpr_32 }
+ - { id: 1, class: vgpr_32 }
+ - { id: 2, class: sreg_64_xexec }
+body: |
+ bb.0:
+ %0 = IMPLICIT_DEF
+ %1 = IMPLICIT_DEF
+ %2 = IMPLICIT_DEF
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ dead $sgpr30_sgpr31 = SI_CALL %2, 0, csr_amdgpu
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ S_ENDPGM 0
+...
+
+# The inits cannot be hoisted into the common dominator because the call in
+# that block clobbers m0.
+
+# GCN-LABEL: name: merge-m0-call-clobber-cross-block
+# GCN: bb.0:
+# GCN-NOT: SI_INIT_M0
+# GCN: SI_CALL
+# GCN: bb.1:
+# GCN: SI_INIT_M0 -1
+# GCN-NEXT: DS_WRITE_B32
+# GCN: bb.2:
+# GCN: SI_INIT_M0 -1
+# GCN-NEXT: DS_WRITE_B32
+
+---
+name: merge-m0-call-clobber-cross-block
+registers:
+ - { id: 0, class: vgpr_32 }
+ - { id: 1, class: vgpr_32 }
+ - { id: 2, class: sreg_64_xexec }
+body: |
+ bb.0:
+ successors: %bb.1, %bb.2
+
+ %0 = IMPLICIT_DEF
+ %1 = IMPLICIT_DEF
+ %2 = IMPLICIT_DEF
+ dead $sgpr30_sgpr31 = SI_CALL %2, 0, csr_amdgpu
+ S_CBRANCH_VCCZ %bb.1, implicit undef $vcc
+ S_BRANCH %bb.2
+
+ bb.1:
+ successors: %bb.3
+
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ S_BRANCH %bb.3
+
+ bb.2:
+ successors: %bb.3
+
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ S_BRANCH %bb.3
+
+ bb.3:
+ S_ENDPGM 0
+...
+
+# Both inits are dominated by the call, so they still merge, but the surviving
+# init must not move above the call.
+
+# GCN-LABEL: name: merge-m0-call-dominates-inits
+# GCN: SI_CALL
+# GCN-NEXT: SI_INIT_M0 -1
+# GCN-NEXT: DS_WRITE_B32
+# GCN-NEXT: DS_WRITE_B32
+# GCN-NEXT: S_ENDPGM
+
+---
+name: merge-m0-call-dominates-inits
+registers:
+ - { id: 0, class: vgpr_32 }
+ - { id: 1, class: vgpr_32 }
+ - { id: 2, class: sreg_64_xexec }
+body: |
+ bb.0:
+ %0 = IMPLICIT_DEF
+ %1 = IMPLICIT_DEF
+ %2 = IMPLICIT_DEF
+ dead $sgpr30_sgpr31 = SI_CALL %2, 0, csr_amdgpu
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ S_ENDPGM 0
+...
+
+# A single init must not be scheduled above a call.
+
+# GCN-LABEL: name: move-m0-call-clobber
+# GCN: SI_CALL
+# GCN-NEXT: SI_INIT_M0 -1
+# GCN-NEXT: DS_WRITE_B32
+# GCN-NEXT: S_ENDPGM
+
+---
+name: move-m0-call-clobber
+registers:
+ - { id: 0, class: vgpr_32 }
+ - { id: 1, class: vgpr_32 }
+ - { id: 2, class: sreg_64_xexec }
+body: |
+ bb.0:
+ %0 = IMPLICIT_DEF
+ %1 = IMPLICIT_DEF
+ %2 = IMPLICIT_DEF
+ dead $sgpr30_sgpr31 = SI_CALL %2, 0, csr_amdgpu
+ SI_INIT_M0 -1, implicit-def $m0
+ DS_WRITE_B32 %0, %1, 0, 0, implicit $m0, implicit $exec
+ S_ENDPGM 0
+...
>From f880fc1097000f5efa1e0c650fc7ee1bb41f8649 Mon Sep 17 00:00:00 2001
From: padivedi <pankajkumar.divedi at amd.com>
Date: Thu, 10 Sep 2026 17:20:51 +0530
Subject: [PATCH 2/2] review: record the call in run to avoid re-scanning
---
llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp | 25 +++++++++++-----------
1 file changed, 12 insertions(+), 13 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
index 7028f89a49f1c..6f1d87e761cee 100644
--- a/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
+++ b/llvm/lib/Target/AMDGPU/SIFixSGPRCopies.cpp
@@ -457,7 +457,7 @@ getFirstNonPrologue(MachineBasicBlock *MBB, const TargetInstrInfo *TII) {
// This is intended to combine M0 initializations, but can work with any
// SGPR. A VGPR cannot be processed since we cannot guarantee vector
// executioon.
-static bool hoistAndMergeSGPRInits(unsigned Reg, MachineFunction &MF,
+static bool hoistAndMergeSGPRInits(unsigned Reg, ArrayRef<MachineInstr *> Calls,
const MachineRegisterInfo &MRI,
const TargetRegisterInfo *TRI,
MachineDominatorTree &MDT,
@@ -469,13 +469,10 @@ static bool hoistAndMergeSGPRInits(unsigned Reg, MachineFunction &MF,
SmallVector<MachineInstr*, 8> Clobbers;
// List of instructions marked for deletion.
SmallPtrSet<MachineInstr *, 8> MergedInstrs;
- // Instructions already classified as an init or a clobber.
- SmallPtrSet<MachineInstr *, 16> Classified;
bool Changed = false;
for (MachineInstr &MI : MRI.def_instructions(Reg)) {
- Classified.insert(&MI);
MachineOperand *Imm = nullptr;
for (MachineOperand &MO : MI.operands()) {
if ((MO.isReg() && ((MO.isDef() && MO.getReg() != Reg) || !MO.isDef())) ||
@@ -492,15 +489,11 @@ static bool hoistAndMergeSGPRInits(unsigned Reg, MachineFunction &MF,
Clobbers.push_back(&MI);
}
- if (Inits.empty())
- return false;
-
// Calls clobber Reg with a regmask operand instead of an explicit def, so
- // they do not appear on the def list of Reg.
- for (MachineBasicBlock &MBB : MF)
- for (MachineInstr &MI : MBB)
- if (!Classified.contains(&MI) && MI.modifiesRegister(Reg, TRI))
- Clobbers.push_back(&MI);
+ // they are not on the def list of Reg.
+ for (MachineInstr *MI : Calls)
+ if (MI->modifiesRegister(Reg, TRI))
+ Clobbers.push_back(MI);
for (auto &Init : Inits) {
auto &Defs = Init.second;
@@ -650,12 +643,18 @@ bool SIFixSGPRCopies::run(MachineFunction &MF) {
// Instructions to re-legalize after changing register classes
SmallVector<MachineInstr *, 8> Relegalize;
+ SmallVector<MachineInstr *, 4> Calls;
for (MachineBasicBlock &MBB : MF) {
for (MachineBasicBlock::iterator I = MBB.begin(), E = MBB.end(); I != E;
++I) {
MachineInstr &MI = *I;
+ // Record calls while walking the function so that hoistAndMergeSGPRInits
+ // can find the registers they clobber through their regmask operand.
+ if (MI.isCall())
+ Calls.push_back(&MI);
+
switch (MI.getOpcode()) {
default:
// scale_src has a register class restricted to low 256 VGPRs, changing
@@ -824,7 +823,7 @@ bool SIFixSGPRCopies::run(MachineFunction &MF) {
TII->legalizeOperands(*Relegalize.pop_back_val(), MDT);
if (MF.getTarget().getOptLevel() > CodeGenOptLevel::None && EnableM0Merge)
- hoistAndMergeSGPRInits(AMDGPU::M0, MF, *MRI, TRI, *MDT, TII);
+ hoistAndMergeSGPRInits(AMDGPU::M0, Calls, *MRI, TRI, *MDT, TII);
SiblingPenalty.clear();
V2SCopies.clear();
More information about the llvm-commits
mailing list