[llvm] [AMDGPU] Fix soft clause hazards missed past the lookahead window (PR #220806)
Arseniy Obolenskiy via llvm-commits
llvm-commits at lists.llvm.org
Fri Sep 11 07:24:20 PDT 2026
https://github.com/aobolensk updated https://github.com/llvm/llvm-project/pull/220806
>From b3ae6613fb170459a83da11f7ef90eccc9bfab66 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Thu, 3 Sep 2026 06:50:22 +0200
Subject: [PATCH 1/3] [AMDGPU] Fix soft clause hazards missed past the
lookahead window
The clause was rebuilt each time from a limited window of past instructions, so hazards further back went undetected
Track the clause as state instead, with no limit
---
.../lib/Target/AMDGPU/GCNHazardRecognizer.cpp | 67 +++++----
llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h | 16 ++-
.../AMDGPU/GlobalISel/vni8-across-blocks.ll | 2 +
.../CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll | 7 +
.../CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll | 3 +
.../AMDGPU/break-smem-soft-clauses.mir | 129 ++++++++++++++++++
.../AMDGPU/break-vmem-soft-clauses.mir | 26 ++++
.../AMDGPU/gfx-callable-return-types.ll | 1 +
llvm/test/CodeGen/AMDGPU/hazard-in-bundle.mir | 22 +++
.../splitkit-getsubrangeformask-phi-extend.ll | 1 +
10 files changed, 244 insertions(+), 30 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
index ee94d1ddd27b5..3c053619869b3 100644
--- a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
@@ -110,6 +110,7 @@ void GCNHazardRecognizer::Reset() {
EmittedInstrs.clear();
EmittedVALUInstrs.clear();
HasPendingWMMACoexecHazard = false;
+ resetClause();
if (isSchedulerMode())
schedulerReset();
}
@@ -676,6 +677,11 @@ void GCNHazardRecognizer::processBundle() {
insertNoopsInBundle(CurrCycleInstr, TII, WaitStates);
}
+ // Use MI, not CurrCycleInstr, which fixHazards may have reset to null.
+ if (WaitStates)
+ resetClause();
+ updateSoftClause(*MI);
+
// It’s unnecessary to track more than MaxLookAhead instructions. Since we
// include the bundled MI directly after, only add a maximum of
// (MaxLookAhead - 1) noops to EmittedInstrs.
@@ -802,6 +808,7 @@ unsigned GCNHazardRecognizer::PreEmitNoopsCommon(MachineInstr *MI) const {
void GCNHazardRecognizer::EmitNoop() {
EmittedInstrs.push_front(nullptr);
+ resetClause();
}
void GCNHazardRecognizer::AdvanceCycle() {
@@ -812,6 +819,10 @@ void GCNHazardRecognizer::AdvanceCycle() {
// emitting any instructions.
if (!CurrCycleInstr) {
EmittedInstrs.push_front(nullptr);
+ // Only reached in scheduler mode, where clause hazards are a heuristic
+ // and stalling cannot break one; reset here or it stalls forever.
+ assert(isSchedulerMode());
+ resetClause();
if (HasPendingWMMACoexecHazard)
EmittedVALUInstrs.push_front(nullptr);
@@ -825,6 +836,8 @@ void GCNHazardRecognizer::AdvanceCycle() {
return;
}
+ updateSoftClause(*CurrCycleInstr);
+
unsigned NumWaitStates = TII.getNumWaitStates(*CurrCycleInstr);
if (!NumWaitStates) {
CurrCycleInstr = nullptr;
@@ -1113,21 +1126,36 @@ static void addRegsToSet(const SIRegisterInfo &TRI,
iterator_range<MachineInstr::const_mop_iterator> Ops,
BitVector &DefSet, BitVector &UseSet) {
for (const MachineOperand &Op : Ops) {
- if (Op.isReg())
+ if (Op.isReg() && Op.getReg().isPhysical())
addRegUnits(TRI, Op.isDef() ? DefSet : UseSet, Op.getReg().asMCReg());
}
}
-void GCNHazardRecognizer::addClauseInst(const MachineInstr &MI) const {
- addRegsToSet(TRI, MI.operands(), ClauseDefs, ClauseUses);
+GCNHazardRecognizer::SoftClauseKind
+GCNHazardRecognizer::getSoftClauseKind(const MachineInstr &MI) {
+ if (SIInstrInfo::isSMRD(MI))
+ return SoftClauseKind::SMEM;
+ if (SIInstrInfo::isVMEM(MI))
+ return SoftClauseKind::VMEM;
+ return SoftClauseKind::None;
}
-static bool breaksSMEMSoftClause(MachineInstr *MI) {
- return !SIInstrInfo::isSMRD(*MI);
-}
+void GCNHazardRecognizer::updateSoftClause(const MachineInstr &MI) {
+ if (!ST.isXNACKEnabled())
+ return;
+
+ // Meta instructions have no encoding, so they neither extend nor break a
+ // clause.
+ if (MI.isMetaInstruction())
+ return;
-static bool breaksVMEMSoftClause(MachineInstr *MI) {
- return !SIInstrInfo::isVMEM(*MI);
+ SoftClauseKind Kind = getSoftClauseKind(MI);
+ if (Kind != ClauseKind) {
+ resetClause();
+ ClauseKind = Kind;
+ }
+ if (Kind != SoftClauseKind::None)
+ addRegsToSet(TRI, MI.operands(), ClauseDefs, ClauseUses);
}
int GCNHazardRecognizer::checkSoftClauseHazards(MachineInstr *MEM) const {
@@ -1136,10 +1164,6 @@ int GCNHazardRecognizer::checkSoftClauseHazards(MachineInstr *MEM) const {
if (!ST.isXNACKEnabled())
return 0;
- bool IsSMRD = TII.isSMRD(*MEM);
-
- resetClause();
-
// A soft-clause is any group of consecutive SMEM instructions. The
// instructions in this group may return out of order and/or may be
// replayed (i.e. the same instruction issued more than once).
@@ -1150,17 +1174,8 @@ int GCNHazardRecognizer::checkSoftClauseHazards(MachineInstr *MEM) const {
// (including itself). If we encounter this situation, we need to break the
// clause by inserting a non SMEM instruction.
- for (MachineInstr *MI : EmittedInstrs) {
- // When we hit a non-SMEM instruction then we have passed the start of the
- // clause and we can stop.
- if (!MI)
- break;
-
- if (IsSMRD ? breaksSMEMSoftClause(MI) : breaksVMEMSoftClause(MI))
- break;
-
- addClauseInst(*MI);
- }
+ if (ClauseKind != getSoftClauseKind(*MEM))
+ return 0;
if (ClauseDefs.none())
return 0;
@@ -1171,11 +1186,11 @@ int GCNHazardRecognizer::checkSoftClauseHazards(MachineInstr *MEM) const {
if (MEM->mayStore())
return 1;
- addClauseInst(*MEM);
-
// If the set of defs and uses intersect then we cannot add this instruction
// to the clause, so we have a hazard.
- return ClauseDefs.anyCommon(ClauseUses) ? 1 : 0;
+ BitVector Defs = ClauseDefs, Uses = ClauseUses;
+ addRegsToSet(TRI, MEM->operands(), Defs, Uses);
+ return Defs.anyCommon(Uses) ? 1 : 0;
}
int GCNHazardRecognizer::checkSMRDHazards(MachineInstr *SMRD) const {
diff --git a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
index 422da557eb47d..168d563a68723 100644
--- a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
+++ b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
@@ -151,18 +151,26 @@ class GCNHazardRecognizer final : public ScheduleHazardRecognizer {
/// Scheduler-mode part of Reset().
void schedulerReset();
+ /// Tracked in emission order because a clause can exceed getMaxLookAhead()
+ /// and so cannot be reconstructed from EmittedInstrs.
+ enum class SoftClauseKind { None, SMEM, VMEM };
+ SoftClauseKind ClauseKind = SoftClauseKind::None;
+
/// RegUnits of uses in the current soft memory clause.
- mutable BitVector ClauseUses;
+ BitVector ClauseUses;
/// RegUnits of defs in the current soft memory clause.
- mutable BitVector ClauseDefs;
+ BitVector ClauseDefs;
- void resetClause() const {
+ void resetClause() {
+ ClauseKind = SoftClauseKind::None;
ClauseUses.reset();
ClauseDefs.reset();
}
- void addClauseInst(const MachineInstr &MI) const;
+ static SoftClauseKind getSoftClauseKind(const MachineInstr &MI);
+
+ void updateSoftClause(const MachineInstr &MI);
/// \returns the number of wait states before another MFMA instruction can be
/// issued after \p MI.
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll
index 9ccbeee37a03a..a21ff31f8c82f 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll
@@ -282,6 +282,7 @@ define amdgpu_kernel void @v256i8_liveout(ptr addrspace(1) %src1, ptr addrspace(
; GFX906-NEXT: global_load_dwordx4 v[53:56], v4, s[0:1] offset:208
; GFX906-NEXT: global_load_dwordx4 v[57:60], v4, s[0:1] offset:224
; GFX906-NEXT: global_load_dwordx4 v[0:3], v4, s[0:1] offset:240
+; GFX906-NEXT: s_nop 0
; GFX906-NEXT: global_load_dwordx4 v[5:8], v4, s[0:1] offset:16
; GFX906-NEXT: global_load_dwordx4 v[9:12], v4, s[0:1] offset:32
; GFX906-NEXT: global_load_dwordx4 v[13:16], v4, s[0:1] offset:48
@@ -309,6 +310,7 @@ define amdgpu_kernel void @v256i8_liveout(ptr addrspace(1) %src1, ptr addrspace(
; GFX906-NEXT: global_load_dwordx4 v[49:52], v4, s[2:3] offset:192
; GFX906-NEXT: global_load_dwordx4 v[53:56], v4, s[2:3] offset:208
; GFX906-NEXT: global_load_dwordx4 v[57:60], v4, s[2:3] offset:224
+; GFX906-NEXT: s_nop 0
; GFX906-NEXT: global_load_dwordx4 v[0:3], v4, s[2:3] offset:240
; GFX906-NEXT: .LBB6_2: ; %bb.2
; GFX906-NEXT: s_or_b64 exec, exec, s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
index 7be6ad4575d69..90bf515d620ea 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
@@ -142895,6 +142895,7 @@ define <64 x bfloat> @bitcast_v128i8_to_v64bf16(<128 x i8> %a, i32 %b) #0 {
; GFX9-NEXT: buffer_load_ushort v60, off, s[0:3], s32 offset:4
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:124
; GFX9-NEXT: buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:372
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -148279,6 +148280,7 @@ define inreg <64 x bfloat> @bitcast_v128i8_to_v64bf16_scalar(<128 x i8> inreg %a
; GFX9-NEXT: buffer_load_ushort v13, off, s[0:3], s32 offset:248
; GFX9-NEXT: buffer_load_ushort v34, off, s[0:3], s32 offset:244
; GFX9-NEXT: buffer_load_ushort v49, off, s[0:3], s32 offset:240
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:236
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: buffer_store_dword v0, off, s[0:3], s32 offset:460 ; 4-byte Folded Spill
@@ -155072,6 +155074,7 @@ define <128 x i8> @bitcast_v64bf16_to_v128i8(<64 x bfloat> %a, i32 %b) #0 {
; GFX9-NEXT: buffer_load_dword v56, off, s[0:3], s32 offset:40 ; 4-byte Folded Reload
; GFX9-NEXT: buffer_load_dword v47, off, s[0:3], s32 offset:44 ; 4-byte Folded Reload
; GFX9-NEXT: buffer_load_dword v46, off, s[0:3], s32 offset:48 ; 4-byte Folded Reload
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_dword v45, off, s[0:3], s32 offset:52 ; 4-byte Folded Reload
; GFX9-NEXT: buffer_load_dword v44, off, s[0:3], s32 offset:56 ; 4-byte Folded Reload
; GFX9-NEXT: buffer_load_dword v43, off, s[0:3], s32 offset:60 ; 4-byte Folded Reload
@@ -169184,6 +169187,7 @@ define <64 x half> @bitcast_v128i8_to_v64f16(<128 x i8> %a, i32 %b) #0 {
; GFX9-NEXT: buffer_load_ushort v60, off, s[0:3], s32 offset:4
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:124
; GFX9-NEXT: buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:372
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -174573,6 +174577,7 @@ define inreg <64 x half> @bitcast_v128i8_to_v64f16_scalar(<128 x i8> inreg %a, i
; GFX9-NEXT: buffer_load_ushort v13, off, s[0:3], s32 offset:248
; GFX9-NEXT: buffer_load_ushort v34, off, s[0:3], s32 offset:244
; GFX9-NEXT: buffer_load_ushort v49, off, s[0:3], s32 offset:240
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:236
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: buffer_store_dword v0, off, s[0:3], s32 offset:460 ; 4-byte Folded Spill
@@ -189889,6 +189894,7 @@ define <64 x i16> @bitcast_v128i8_to_v64i16(<128 x i8> %a, i32 %b) #0 {
; GFX9-NEXT: buffer_load_ushort v60, off, s[0:3], s32 offset:4
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:124
; GFX9-NEXT: buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:372
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -195278,6 +195284,7 @@ define inreg <64 x i16> @bitcast_v128i8_to_v64i16_scalar(<128 x i8> inreg %a, i3
; GFX9-NEXT: buffer_load_ushort v13, off, s[0:3], s32 offset:248
; GFX9-NEXT: buffer_load_ushort v34, off, s[0:3], s32 offset:244
; GFX9-NEXT: buffer_load_ushort v49, off, s[0:3], s32 offset:240
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:236
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: buffer_store_dword v0, off, s[0:3], s32 offset:460 ; 4-byte Folded Spill
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
index 280a0f17a4bc8..e29b023703a59 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
@@ -69258,6 +69258,7 @@ define <32 x i16> @bitcast_v64i8_to_v32i16(<64 x i8> %a, i32 %b) #0 {
; GFX9-NEXT: buffer_store_dword v14, off, s[0:3], s32 offset:252 ; 4-byte Folded Spill
; GFX9-NEXT: buffer_store_dword v13, off, s[0:3], s32 offset:256 ; 4-byte Folded Spill
; GFX9-NEXT: buffer_load_ushort v48, off, s[0:3], s32 offset:36
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v60, off, s[0:3], s32 offset:28
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:24
; GFX9-NEXT: buffer_load_ushort v63, off, s[0:3], s32 offset:20
@@ -81995,6 +81996,7 @@ define <32 x half> @bitcast_v64i8_to_v32f16(<64 x i8> %a, i32 %b) #0 {
; GFX9-NEXT: buffer_store_dword v14, off, s[0:3], s32 offset:252 ; 4-byte Folded Spill
; GFX9-NEXT: buffer_store_dword v13, off, s[0:3], s32 offset:256 ; 4-byte Folded Spill
; GFX9-NEXT: buffer_load_ushort v48, off, s[0:3], s32 offset:36
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v60, off, s[0:3], s32 offset:28
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:24
; GFX9-NEXT: buffer_load_ushort v63, off, s[0:3], s32 offset:20
@@ -92665,6 +92667,7 @@ define <32 x bfloat> @bitcast_v64i8_to_v32bf16(<64 x i8> %a, i32 %b) #0 {
; GFX9-NEXT: buffer_store_dword v14, off, s[0:3], s32 offset:252 ; 4-byte Folded Spill
; GFX9-NEXT: buffer_store_dword v13, off, s[0:3], s32 offset:256 ; 4-byte Folded Spill
; GFX9-NEXT: buffer_load_ushort v48, off, s[0:3], s32 offset:36
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v60, off, s[0:3], s32 offset:28
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:24
; GFX9-NEXT: buffer_load_ushort v63, off, s[0:3], s32 offset:20
diff --git a/llvm/test/CodeGen/AMDGPU/break-smem-soft-clauses.mir b/llvm/test/CodeGen/AMDGPU/break-smem-soft-clauses.mir
index 8c5ef9cd5525b..b9e4a48aec961 100644
--- a/llvm/test/CodeGen/AMDGPU/break-smem-soft-clauses.mir
+++ b/llvm/test/CodeGen/AMDGPU/break-smem-soft-clauses.mir
@@ -351,3 +351,132 @@ body: |
$sgpr12_sgpr13 = S_LOAD_DWORDX2_IMM $sgpr6_sgpr7, 0, 0
S_ENDPGM 0
...
+---
+# WAR at clause distance 6. This is beyond the scheduler lookahead window
+# (MaxLookAhead = 5), so the clause must be tracked statefully rather than
+# reconstructed from the emitted-instruction window.
+name: break_smem_clause_war_distance_6
+
+body: |
+ bb.0:
+ ; GCN-LABEL: name: break_smem_clause_war_distance_6
+ ; GCN: $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+ ; GCN-NEXT: $sgpr1 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+ ; GCN-NEXT: $sgpr2 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+ ; GCN-NEXT: $sgpr3 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+ ; GCN-NEXT: $sgpr4 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+ ; GCN-NEXT: $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+ ; XNACK-NEXT: S_NOP 0
+ ; GCN-NEXT: $sgpr10 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+ ; GCN-NEXT: S_ENDPGM 0
+ $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+ $sgpr1 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+ $sgpr2 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+ $sgpr3 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+ $sgpr4 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+ $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+ $sgpr10 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+ S_ENDPGM 0
+...
+---
+# Same as above without the conflict: a long clause on its own must not get a
+# nop.
+name: smem_clause_distance_6_no_conflict
+
+body: |
+ bb.0:
+ ; GCN-LABEL: name: smem_clause_distance_6_no_conflict
+ ; GCN: $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+ ; GCN-NEXT: $sgpr1 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+ ; GCN-NEXT: $sgpr2 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+ ; GCN-NEXT: $sgpr3 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+ ; GCN-NEXT: $sgpr4 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+ ; GCN-NEXT: $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+ ; GCN-NEXT: $sgpr6 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+ ; GCN-NEXT: S_ENDPGM 0
+ $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+ $sgpr1 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+ $sgpr2 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+ $sgpr3 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+ $sgpr4 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+ $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+ $sgpr6 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+ S_ENDPGM 0
+...
+---
+# KILL is a meta instruction with no encoding, so it does not break the
+# hardware clause and the WAR across it must still be detected.
+name: kill_does_not_break_smem_clause
+
+body: |
+ bb.0:
+ ; GCN-LABEL: name: kill_does_not_break_smem_clause
+ ; GCN: $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+ ; GCN-NEXT: KILL $sgpr8_sgpr9
+ ; XNACK-NEXT: S_NOP 0
+ ; GCN-NEXT: $sgpr10 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+ ; GCN-NEXT: S_ENDPGM 0
+ $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+ KILL $sgpr8_sgpr9
+ $sgpr10 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+ S_ENDPGM 0
+...
+---
+# A KILL's operands are not part of the clause: overwriting them is fine.
+name: kill_operand_does_not_join_smem_clause
+
+body: |
+ bb.0:
+ ; GCN-LABEL: name: kill_operand_does_not_join_smem_clause
+ ; GCN: $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+ ; GCN-NEXT: KILL $sgpr8_sgpr9
+ ; GCN-NEXT: $sgpr8 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+ ; GCN-NEXT: S_ENDPGM 0
+ $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+ KILL $sgpr8_sgpr9
+ $sgpr8 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+ S_ENDPGM 0
+...
+---
+# The shape SIFormMemoryClauses leaves at its 15-instruction limit: a run of
+# loads, the KILLs that ended the first bundle, then more loads. The KILLs do
+# not break the hardware clause, and load 17 overwrites load 1's address.
+name: break_smem_clause_kill_boundary_17_loads
+
+body: |
+ bb.0:
+ ; GCN-LABEL: name: break_smem_clause_kill_boundary_17_loads
+ ; GCN: $sgpr40 = S_LOAD_DWORD_IMM $sgpr2_sgpr3, 0, 0
+ ; GCN: $sgpr51 = S_LOAD_DWORD_IMM $sgpr24_sgpr25, 0, 0
+ ; GCN-NEXT: KILL killed renamable $sgpr2_sgpr3
+ ; GCN-NEXT: KILL killed renamable $sgpr4_sgpr5
+ ; GCN-NEXT: KILL killed renamable $sgpr6_sgpr7
+ ; GCN-NEXT: $sgpr52 = S_LOAD_DWORD_IMM $sgpr26_sgpr27, 0, 0
+ ; GCN-NEXT: $sgpr53 = S_LOAD_DWORD_IMM $sgpr28_sgpr29, 0, 0
+ ; GCN-NEXT: $sgpr54 = S_LOAD_DWORD_IMM $sgpr30_sgpr31, 0, 0
+ ; GCN-NEXT: $sgpr55 = S_LOAD_DWORD_IMM $sgpr32_sgpr33, 0, 0
+ ; XNACK-NEXT: S_NOP 0
+ ; GCN-NEXT: $sgpr2 = S_LOAD_DWORD_IMM $sgpr34_sgpr35, 0, 0
+ ; GCN-NEXT: S_ENDPGM 0
+ $sgpr40 = S_LOAD_DWORD_IMM $sgpr2_sgpr3, 0, 0
+ $sgpr41 = S_LOAD_DWORD_IMM $sgpr4_sgpr5, 0, 0
+ $sgpr42 = S_LOAD_DWORD_IMM $sgpr6_sgpr7, 0, 0
+ $sgpr43 = S_LOAD_DWORD_IMM $sgpr8_sgpr9, 0, 0
+ $sgpr44 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+ $sgpr45 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+ $sgpr46 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+ $sgpr47 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+ $sgpr48 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+ $sgpr49 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+ $sgpr50 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+ $sgpr51 = S_LOAD_DWORD_IMM $sgpr24_sgpr25, 0, 0
+ KILL killed renamable $sgpr2_sgpr3
+ KILL killed renamable $sgpr4_sgpr5
+ KILL killed renamable $sgpr6_sgpr7
+ $sgpr52 = S_LOAD_DWORD_IMM $sgpr26_sgpr27, 0, 0
+ $sgpr53 = S_LOAD_DWORD_IMM $sgpr28_sgpr29, 0, 0
+ $sgpr54 = S_LOAD_DWORD_IMM $sgpr30_sgpr31, 0, 0
+ $sgpr55 = S_LOAD_DWORD_IMM $sgpr32_sgpr33, 0, 0
+ $sgpr2 = S_LOAD_DWORD_IMM $sgpr34_sgpr35, 0, 0
+ S_ENDPGM 0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/break-vmem-soft-clauses.mir b/llvm/test/CodeGen/AMDGPU/break-vmem-soft-clauses.mir
index 55460fd0d6d20..9c6a0a5554424 100644
--- a/llvm/test/CodeGen/AMDGPU/break-vmem-soft-clauses.mir
+++ b/llvm/test/CodeGen/AMDGPU/break-vmem-soft-clauses.mir
@@ -577,3 +577,29 @@ body: |
$vgpr11 = FLAT_LOAD_DWORD $vgpr4_vgpr5, 0, 0, implicit $exec, implicit $flat_scr
S_ENDPGM 0
...
+---
+# WAR at clause distance 6, beyond the scheduler lookahead window
+# (MaxLookAhead = 5).
+name: break_vmem_clause_war_distance_6
+
+body: |
+ bb.0:
+ ; GCN-LABEL: name: break_vmem_clause_war_distance_6
+ ; GCN: $vgpr0 = FLAT_LOAD_DWORD $vgpr2_vgpr3, 0, 0, implicit $exec, implicit $flat_scr
+ ; GCN-NEXT: $vgpr1 = FLAT_LOAD_DWORD $vgpr4_vgpr5, 0, 0, implicit $exec, implicit $flat_scr
+ ; GCN-NEXT: $vgpr10 = FLAT_LOAD_DWORD $vgpr6_vgpr7, 0, 0, implicit $exec, implicit $flat_scr
+ ; GCN-NEXT: $vgpr11 = FLAT_LOAD_DWORD $vgpr8_vgpr9, 0, 0, implicit $exec, implicit $flat_scr
+ ; GCN-NEXT: $vgpr12 = FLAT_LOAD_DWORD $vgpr14_vgpr15, 0, 0, implicit $exec, implicit $flat_scr
+ ; GCN-NEXT: $vgpr13 = FLAT_LOAD_DWORD $vgpr16_vgpr17, 0, 0, implicit $exec, implicit $flat_scr
+ ; XNACK-NEXT: S_NOP 0
+ ; GCN-NEXT: $vgpr2 = FLAT_LOAD_DWORD $vgpr18_vgpr19, 0, 0, implicit $exec, implicit $flat_scr
+ ; GCN-NEXT: S_ENDPGM 0
+ $vgpr0 = FLAT_LOAD_DWORD $vgpr2_vgpr3, 0, 0, implicit $exec, implicit $flat_scr
+ $vgpr1 = FLAT_LOAD_DWORD $vgpr4_vgpr5, 0, 0, implicit $exec, implicit $flat_scr
+ $vgpr10 = FLAT_LOAD_DWORD $vgpr6_vgpr7, 0, 0, implicit $exec, implicit $flat_scr
+ $vgpr11 = FLAT_LOAD_DWORD $vgpr8_vgpr9, 0, 0, implicit $exec, implicit $flat_scr
+ $vgpr12 = FLAT_LOAD_DWORD $vgpr14_vgpr15, 0, 0, implicit $exec, implicit $flat_scr
+ $vgpr13 = FLAT_LOAD_DWORD $vgpr16_vgpr17, 0, 0, implicit $exec, implicit $flat_scr
+ $vgpr2 = FLAT_LOAD_DWORD $vgpr18_vgpr19, 0, 0, implicit $exec, implicit $flat_scr
+ S_ENDPGM 0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/gfx-callable-return-types.ll b/llvm/test/CodeGen/AMDGPU/gfx-callable-return-types.ll
index f805b11d35db5..782df520b817d 100644
--- a/llvm/test/CodeGen/AMDGPU/gfx-callable-return-types.ll
+++ b/llvm/test/CodeGen/AMDGPU/gfx-callable-return-types.ll
@@ -2900,6 +2900,7 @@ define amdgpu_gfx void @call_72xi32() #1 {
; GFX9-NEXT: buffer_store_dword v7, off, s[0:3], s32 offset:156
; GFX9-NEXT: buffer_store_dword v8, off, s[0:3], s32 offset:160
; GFX9-NEXT: buffer_load_dword v2, off, s[0:3], s33 offset:1536 ; 4-byte Folded Reload
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_dword v3, off, s[0:3], s33 offset:1540 ; 4-byte Folded Reload
; GFX9-NEXT: buffer_load_dword v4, off, s[0:3], s33 offset:1544 ; 4-byte Folded Reload
; GFX9-NEXT: buffer_load_dword v5, off, s[0:3], s33 offset:1548 ; 4-byte Folded Reload
diff --git a/llvm/test/CodeGen/AMDGPU/hazard-in-bundle.mir b/llvm/test/CodeGen/AMDGPU/hazard-in-bundle.mir
index 3cc5d455d36d8..0a3477f625db4 100644
--- a/llvm/test/CodeGen/AMDGPU/hazard-in-bundle.mir
+++ b/llvm/test/CodeGen/AMDGPU/hazard-in-bundle.mir
@@ -82,3 +82,25 @@ body: |
}
S_ENDPGM 0
...
+
+# GCN-LABEL: name: break_smem_clause_beyond_max_look_ahead_in_bundle
+# GCN: $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+# XNACK-NEXT: S_NOP
+# NOXNACK-NOT: S_NOP
+# GCN-NEXT: $sgpr10 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+# GCN: }
+---
+name: break_smem_clause_beyond_max_look_ahead_in_bundle
+body: |
+ bb.0:
+ BUNDLE implicit-def $sgpr0, implicit-def $sgpr1, implicit-def $sgpr2, implicit-def $sgpr3, implicit-def $sgpr4, implicit-def $sgpr5, implicit-def $sgpr10, implicit $sgpr10_sgpr11, implicit $sgpr12_sgpr13, implicit $sgpr14_sgpr15, implicit $sgpr16_sgpr17, implicit $sgpr18_sgpr19, implicit $sgpr20_sgpr21, implicit $sgpr22_sgpr23 {
+ $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+ $sgpr1 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+ $sgpr2 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+ $sgpr3 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+ $sgpr4 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+ $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+ $sgpr10 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+ }
+ S_ENDPGM 0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll b/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll
index ab946d8667f8c..87692260e90e2 100644
--- a/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll
+++ b/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll
@@ -299,6 +299,7 @@ define void @f(ptr %p, <16 x i1> %m, <16 x i63> %pt, <16 x i1> %sc,
; CHECK-NEXT: buffer_load_dword a9, off, s[0:3], s32 offset:456
; CHECK-NEXT: buffer_load_dword v12, off, s[0:3], s32 offset:8
; CHECK-NEXT: buffer_load_dword v10, off, s[0:3], s32 offset:4
+; CHECK-NEXT: s_nop 0
; CHECK-NEXT: buffer_load_dword v6, off, s[0:3], s32
; CHECK-NEXT: buffer_load_dword v24, off, s[0:3], s32 offset:708
; CHECK-NEXT: buffer_load_dword v56, off, s[0:3], s32 offset:704
>From a3bd0b2c9e937e268055cdbb2829db4ee1a1fdee Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Fri, 11 Sep 2026 11:01:14 +0200
Subject: [PATCH 2/3] upd
---
llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll | 13 +++++++++++++
1 file changed, 13 insertions(+)
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
index 62ad9928797e2..b3e2f6ef88fab 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
@@ -20107,6 +20107,7 @@ define inreg <32 x i32> @bitcast_v128i8_to_v32i32_scalar(<128 x i8> inreg %a, i3
; GFX9-NEXT: buffer_load_ushort v25, off, s[0:3], s32 offset:140
; GFX9-NEXT: buffer_load_ushort v24, off, s[0:3], s32 offset:136
; GFX9-NEXT: buffer_load_ushort v23, off, s[0:3], s32 offset:132
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:128
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:124
; GFX9-NEXT: buffer_load_ushort v18, off, s[0:3], s32 offset:120
@@ -56225,6 +56226,7 @@ define inreg <32 x float> @bitcast_v128i8_to_v32f32_scalar(<128 x i8> inreg %a,
; GFX9-NEXT: buffer_load_ushort v25, off, s[0:3], s32 offset:140
; GFX9-NEXT: buffer_load_ushort v24, off, s[0:3], s32 offset:136
; GFX9-NEXT: buffer_load_ushort v23, off, s[0:3], s32 offset:132
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:128
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:124
; GFX9-NEXT: buffer_load_ushort v18, off, s[0:3], s32 offset:120
@@ -90562,6 +90564,7 @@ define inreg <16 x i64> @bitcast_v128i8_to_v16i64_scalar(<128 x i8> inreg %a, i3
; GFX9-NEXT: buffer_load_ushort v25, off, s[0:3], s32 offset:140
; GFX9-NEXT: buffer_load_ushort v24, off, s[0:3], s32 offset:136
; GFX9-NEXT: buffer_load_ushort v23, off, s[0:3], s32 offset:132
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:128
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:124
; GFX9-NEXT: buffer_load_ushort v18, off, s[0:3], s32 offset:120
@@ -123799,6 +123802,7 @@ define inreg <16 x double> @bitcast_v128i8_to_v16f64_scalar(<128 x i8> inreg %a,
; GFX9-NEXT: buffer_load_ushort v25, off, s[0:3], s32 offset:140
; GFX9-NEXT: buffer_load_ushort v24, off, s[0:3], s32 offset:136
; GFX9-NEXT: buffer_load_ushort v23, off, s[0:3], s32 offset:132
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:128
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:124
; GFX9-NEXT: buffer_load_ushort v18, off, s[0:3], s32 offset:120
@@ -143855,6 +143859,7 @@ define <64 x bfloat> @bitcast_v128i8_to_v64bf16(<128 x i8> %a, i32 %b) #0 {
; GFX9-NEXT: buffer_load_ushort v60, off, s[0:3], s32 offset:4
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:124
; GFX9-NEXT: buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:372
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -149251,6 +149256,7 @@ define inreg <64 x bfloat> @bitcast_v128i8_to_v64bf16_scalar(<128 x i8> inreg %a
; GFX9-NEXT: buffer_load_ushort v6, off, s[0:3], s32 offset:272
; GFX9-NEXT: buffer_load_ushort v8, off, s[0:3], s32 offset:268
; GFX9-NEXT: buffer_load_ushort v41, off, s[0:3], s32 offset:264
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v2, off, s[0:3], s32 offset:260
; GFX9-NEXT: buffer_load_ushort v24, off, s[0:3], s32 offset:256
; GFX9-NEXT: buffer_load_ushort v9, off, s[0:3], s32 offset:252
@@ -149348,6 +149354,7 @@ define inreg <64 x bfloat> @bitcast_v128i8_to_v64bf16_scalar(<128 x i8> inreg %a
; GFX9-NEXT: buffer_load_ushort v30, off, s[0:3], s32 offset:60
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:56
; GFX9-NEXT: buffer_load_ushort v52, off, s[0:3], s32 offset:52
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v14, off, s[0:3], s32 offset:48
; GFX9-NEXT: buffer_load_ushort v29, off, s[0:3], s32 offset:44
; GFX9-NEXT: buffer_load_ushort v39, off, s[0:3], s32 offset:40
@@ -170343,6 +170350,7 @@ define <64 x half> @bitcast_v128i8_to_v64f16(<128 x i8> %a, i32 %b) #0 {
; GFX9-NEXT: buffer_load_ushort v60, off, s[0:3], s32 offset:4
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:124
; GFX9-NEXT: buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:372
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -175739,6 +175747,7 @@ define inreg <64 x half> @bitcast_v128i8_to_v64f16_scalar(<128 x i8> inreg %a, i
; GFX9-NEXT: buffer_load_ushort v6, off, s[0:3], s32 offset:272
; GFX9-NEXT: buffer_load_ushort v8, off, s[0:3], s32 offset:268
; GFX9-NEXT: buffer_load_ushort v41, off, s[0:3], s32 offset:264
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v2, off, s[0:3], s32 offset:260
; GFX9-NEXT: buffer_load_ushort v24, off, s[0:3], s32 offset:256
; GFX9-NEXT: buffer_load_ushort v9, off, s[0:3], s32 offset:252
@@ -175836,6 +175845,7 @@ define inreg <64 x half> @bitcast_v128i8_to_v64f16_scalar(<128 x i8> inreg %a, i
; GFX9-NEXT: buffer_load_ushort v30, off, s[0:3], s32 offset:60
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:56
; GFX9-NEXT: buffer_load_ushort v52, off, s[0:3], s32 offset:52
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v14, off, s[0:3], s32 offset:48
; GFX9-NEXT: buffer_load_ushort v29, off, s[0:3], s32 offset:44
; GFX9-NEXT: buffer_load_ushort v39, off, s[0:3], s32 offset:40
@@ -191200,6 +191210,7 @@ define <64 x i16> @bitcast_v128i8_to_v64i16(<128 x i8> %a, i32 %b) #0 {
; GFX9-NEXT: buffer_load_ushort v60, off, s[0:3], s32 offset:4
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:124
; GFX9-NEXT: buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v0, off, s[0:3], s32 offset:372
; GFX9-NEXT: s_waitcnt vmcnt(0)
; GFX9-NEXT: buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -196596,6 +196607,7 @@ define inreg <64 x i16> @bitcast_v128i8_to_v64i16_scalar(<128 x i8> inreg %a, i3
; GFX9-NEXT: buffer_load_ushort v6, off, s[0:3], s32 offset:272
; GFX9-NEXT: buffer_load_ushort v8, off, s[0:3], s32 offset:268
; GFX9-NEXT: buffer_load_ushort v41, off, s[0:3], s32 offset:264
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v2, off, s[0:3], s32 offset:260
; GFX9-NEXT: buffer_load_ushort v24, off, s[0:3], s32 offset:256
; GFX9-NEXT: buffer_load_ushort v9, off, s[0:3], s32 offset:252
@@ -196693,6 +196705,7 @@ define inreg <64 x i16> @bitcast_v128i8_to_v64i16_scalar(<128 x i8> inreg %a, i3
; GFX9-NEXT: buffer_load_ushort v30, off, s[0:3], s32 offset:60
; GFX9-NEXT: buffer_load_ushort v62, off, s[0:3], s32 offset:56
; GFX9-NEXT: buffer_load_ushort v52, off, s[0:3], s32 offset:52
+; GFX9-NEXT: s_nop 0
; GFX9-NEXT: buffer_load_ushort v14, off, s[0:3], s32 offset:48
; GFX9-NEXT: buffer_load_ushort v29, off, s[0:3], s32 offset:44
; GFX9-NEXT: buffer_load_ushort v39, off, s[0:3], s32 offset:40
>From 136fa40d20b4fd40d5ebc1df2a9c3bb3dddd5125 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Fri, 11 Sep 2026 16:23:57 +0200
Subject: [PATCH 3/3] comnments
---
.../lib/Target/AMDGPU/GCNHazardRecognizer.cpp | 45 ++++++++++++++-----
llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h | 15 ++++---
2 files changed, 42 insertions(+), 18 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
index 1edd04293a5db..083fe6ac109bc 100644
--- a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
@@ -679,12 +679,12 @@ void GCNHazardRecognizer::processBundle() {
unsigned WaitStates = PreEmitNoopsCommon(CurrCycleInstr);
if (isHazardRecognizerMode()) {
+ // fixHazards can reset CurrCycleInstr to null, so use MI from here on.
fixHazards(CurrCycleInstr);
- insertNoopsInBundle(CurrCycleInstr, TII, WaitStates);
+ insertNoopsInBundle(&*MI, TII, WaitStates);
}
- // Use MI, not CurrCycleInstr, which fixHazards may have reset to null.
if (WaitStates)
resetClause();
updateSoftClause(*MI);
@@ -695,7 +695,7 @@ void GCNHazardRecognizer::processBundle() {
for (unsigned i = 0, e = std::min(WaitStates, MaxLookAhead - 1); i < e; ++i)
EmittedInstrs.push_front(nullptr);
- EmittedInstrs.push_front(CurrCycleInstr);
+ EmittedInstrs.push_front(&*MI);
EmittedInstrs.resize(MaxLookAhead);
}
CurrCycleInstr = nullptr;
@@ -825,10 +825,10 @@ void GCNHazardRecognizer::AdvanceCycle() {
// When the scheduler detects a stall, it will call AdvanceCycle() without
// emitting any instructions.
if (!CurrCycleInstr) {
+ assert(isSchedulerMode() && "stall cycles only occur in scheduler mode");
EmittedInstrs.push_front(nullptr);
- // Only reached in scheduler mode, where clause hazards are a heuristic
- // and stalling cannot break one; reset here or it stalls forever.
- assert(isSchedulerMode());
+ // A stall does not really break a clause, but model it as one or the
+ // scheduler stalls on the same hazard forever.
resetClause();
if (HasPendingWMMACoexecHazard)
@@ -1138,6 +1138,13 @@ static void addRegsToSet(const SIRegisterInfo &TRI,
}
}
+static bool anyRegUnitSet(const SIRegisterInfo &TRI, const BitVector &Units,
+ MCRegister Reg) {
+ return any_of(TRI.regunits(Reg), [&Units](MCRegUnit Unit) {
+ return Units.test(static_cast<unsigned>(Unit));
+ });
+}
+
GCNHazardRecognizer::SoftClauseKind
GCNHazardRecognizer::getSoftClauseKind(const MachineInstr &MI) {
if (SIInstrInfo::isSMRD(MI))
@@ -1148,10 +1155,11 @@ GCNHazardRecognizer::getSoftClauseKind(const MachineInstr &MI) {
}
void GCNHazardRecognizer::updateSoftClause(const MachineInstr &MI) {
- if (!ST.isXNACKEnabled())
+ // checkSoftClauseHazards is only reached when hasPhysRegs().
+ if (!hasPhysRegs() || !ST.isXNACKEnabled())
return;
- // Meta instructions have no encoding, so they neither extend nor break a
+ // Meta instructions have no encoding, so they cannot break or extend a
// clause.
if (MI.isMetaInstruction())
return;
@@ -1194,10 +1202,23 @@ int GCNHazardRecognizer::checkSoftClauseHazards(MachineInstr *MEM) const {
return 1;
// If the set of defs and uses intersect then we cannot add this instruction
- // to the clause, so we have a hazard.
- BitVector Defs = ClauseDefs, Uses = ClauseUses;
- addRegsToSet(TRI, MEM->operands(), Defs, Uses);
- return Defs.anyCommon(Uses) ? 1 : 0;
+ // to the clause, so we have a hazard. Test MEM in place to avoid copying a
+ // reg-unit-sized BitVector per query.
+ if (ClauseDefs.anyCommon(ClauseUses))
+ return 1;
+
+ for (const MachineOperand &Op : MEM->operands()) {
+ if (!Op.isReg() || !Op.getReg().isPhysical())
+ continue;
+ MCRegister Reg = Op.getReg().asMCReg();
+ if (anyRegUnitSet(TRI, Op.isDef() ? ClauseUses : ClauseDefs, Reg))
+ return 1;
+ // MEM can also conflict with itself.
+ if (Op.isDef() && MEM->readsRegister(Reg, &TRI))
+ return 1;
+ }
+
+ return 0;
}
int GCNHazardRecognizer::checkSMRDHazards(MachineInstr *SMRD) const {
diff --git a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
index 168d563a68723..4cd417eb3bb2c 100644
--- a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
+++ b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
@@ -151,18 +151,21 @@ class GCNHazardRecognizer final : public ScheduleHazardRecognizer {
/// Scheduler-mode part of Reset().
void schedulerReset();
- /// Tracked in emission order because a clause can exceed getMaxLookAhead()
- /// and so cannot be reconstructed from EmittedInstrs.
enum class SoftClauseKind { None, SMEM, VMEM };
- SoftClauseKind ClauseKind = SoftClauseKind::None;
- /// RegUnits of uses in the current soft memory clause.
+ /// The current soft memory clause, as reg units. It cannot be rebuilt from
+ /// EmittedInstrs: getMaxLookAhead() bounds wait states, not clause length.
+ SoftClauseKind ClauseKind = SoftClauseKind::None;
BitVector ClauseUses;
-
- /// RegUnits of defs in the current soft memory clause.
BitVector ClauseDefs;
void resetClause() {
+ // EmitNoops() calls this once per nop, and clearing is not free.
+ if (ClauseKind == SoftClauseKind::None) {
+ assert(ClauseUses.none() && ClauseDefs.none() &&
+ "no clause kind implies no tracked reg units");
+ return;
+ }
ClauseKind = SoftClauseKind::None;
ClauseUses.reset();
ClauseDefs.reset();
More information about the llvm-commits
mailing list