[llvm] [AMDGPU] Fix soft clause hazards missed past the lookahead window (PR #220806)

Arseniy Obolenskiy via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 11 07:24:20 PDT 2026


https://github.com/aobolensk updated https://github.com/llvm/llvm-project/pull/220806

>From b3ae6613fb170459a83da11f7ef90eccc9bfab66 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Thu, 3 Sep 2026 06:50:22 +0200
Subject: [PATCH 1/3] [AMDGPU] Fix soft clause hazards missed past the
 lookahead window

The clause was rebuilt each time from a limited window of past instructions, so hazards further back went undetected

Track the clause as state instead, with no limit
---
 .../lib/Target/AMDGPU/GCNHazardRecognizer.cpp |  67 +++++----
 llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h  |  16 ++-
 .../AMDGPU/GlobalISel/vni8-across-blocks.ll   |   2 +
 .../CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll  |   7 +
 .../CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll   |   3 +
 .../AMDGPU/break-smem-soft-clauses.mir        | 129 ++++++++++++++++++
 .../AMDGPU/break-vmem-soft-clauses.mir        |  26 ++++
 .../AMDGPU/gfx-callable-return-types.ll       |   1 +
 llvm/test/CodeGen/AMDGPU/hazard-in-bundle.mir |  22 +++
 .../splitkit-getsubrangeformask-phi-extend.ll |   1 +
 10 files changed, 244 insertions(+), 30 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
index ee94d1ddd27b5..3c053619869b3 100644
--- a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
@@ -110,6 +110,7 @@ void GCNHazardRecognizer::Reset() {
   EmittedInstrs.clear();
   EmittedVALUInstrs.clear();
   HasPendingWMMACoexecHazard = false;
+  resetClause();
   if (isSchedulerMode())
     schedulerReset();
 }
@@ -676,6 +677,11 @@ void GCNHazardRecognizer::processBundle() {
       insertNoopsInBundle(CurrCycleInstr, TII, WaitStates);
     }
 
+    // Use MI, not CurrCycleInstr, which fixHazards may have reset to null.
+    if (WaitStates)
+      resetClause();
+    updateSoftClause(*MI);
+
     // It’s unnecessary to track more than MaxLookAhead instructions. Since we
     // include the bundled MI directly after, only add a maximum of
     // (MaxLookAhead - 1) noops to EmittedInstrs.
@@ -802,6 +808,7 @@ unsigned GCNHazardRecognizer::PreEmitNoopsCommon(MachineInstr *MI) const {
 
 void GCNHazardRecognizer::EmitNoop() {
   EmittedInstrs.push_front(nullptr);
+  resetClause();
 }
 
 void GCNHazardRecognizer::AdvanceCycle() {
@@ -812,6 +819,10 @@ void GCNHazardRecognizer::AdvanceCycle() {
   // emitting any instructions.
   if (!CurrCycleInstr) {
     EmittedInstrs.push_front(nullptr);
+    // Only reached in scheduler mode, where clause hazards are a heuristic
+    // and stalling cannot break one; reset here or it stalls forever.
+    assert(isSchedulerMode());
+    resetClause();
 
     if (HasPendingWMMACoexecHazard)
       EmittedVALUInstrs.push_front(nullptr);
@@ -825,6 +836,8 @@ void GCNHazardRecognizer::AdvanceCycle() {
     return;
   }
 
+  updateSoftClause(*CurrCycleInstr);
+
   unsigned NumWaitStates = TII.getNumWaitStates(*CurrCycleInstr);
   if (!NumWaitStates) {
     CurrCycleInstr = nullptr;
@@ -1113,21 +1126,36 @@ static void addRegsToSet(const SIRegisterInfo &TRI,
                          iterator_range<MachineInstr::const_mop_iterator> Ops,
                          BitVector &DefSet, BitVector &UseSet) {
   for (const MachineOperand &Op : Ops) {
-    if (Op.isReg())
+    if (Op.isReg() && Op.getReg().isPhysical())
       addRegUnits(TRI, Op.isDef() ? DefSet : UseSet, Op.getReg().asMCReg());
   }
 }
 
-void GCNHazardRecognizer::addClauseInst(const MachineInstr &MI) const {
-  addRegsToSet(TRI, MI.operands(), ClauseDefs, ClauseUses);
+GCNHazardRecognizer::SoftClauseKind
+GCNHazardRecognizer::getSoftClauseKind(const MachineInstr &MI) {
+  if (SIInstrInfo::isSMRD(MI))
+    return SoftClauseKind::SMEM;
+  if (SIInstrInfo::isVMEM(MI))
+    return SoftClauseKind::VMEM;
+  return SoftClauseKind::None;
 }
 
-static bool breaksSMEMSoftClause(MachineInstr *MI) {
-  return !SIInstrInfo::isSMRD(*MI);
-}
+void GCNHazardRecognizer::updateSoftClause(const MachineInstr &MI) {
+  if (!ST.isXNACKEnabled())
+    return;
+
+  // Meta instructions have no encoding, so they neither extend nor break a
+  // clause.
+  if (MI.isMetaInstruction())
+    return;
 
-static bool breaksVMEMSoftClause(MachineInstr *MI) {
-  return !SIInstrInfo::isVMEM(*MI);
+  SoftClauseKind Kind = getSoftClauseKind(MI);
+  if (Kind != ClauseKind) {
+    resetClause();
+    ClauseKind = Kind;
+  }
+  if (Kind != SoftClauseKind::None)
+    addRegsToSet(TRI, MI.operands(), ClauseDefs, ClauseUses);
 }
 
 int GCNHazardRecognizer::checkSoftClauseHazards(MachineInstr *MEM) const {
@@ -1136,10 +1164,6 @@ int GCNHazardRecognizer::checkSoftClauseHazards(MachineInstr *MEM) const {
   if (!ST.isXNACKEnabled())
     return 0;
 
-  bool IsSMRD = TII.isSMRD(*MEM);
-
-  resetClause();
-
   // A soft-clause is any group of consecutive SMEM instructions.  The
   // instructions in this group may return out of order and/or may be
   // replayed (i.e. the same instruction issued more than once).
@@ -1150,17 +1174,8 @@ int GCNHazardRecognizer::checkSoftClauseHazards(MachineInstr *MEM) const {
   // (including itself). If we encounter this situation, we need to break the
   // clause by inserting a non SMEM instruction.
 
-  for (MachineInstr *MI : EmittedInstrs) {
-    // When we hit a non-SMEM instruction then we have passed the start of the
-    // clause and we can stop.
-    if (!MI)
-      break;
-
-    if (IsSMRD ? breaksSMEMSoftClause(MI) : breaksVMEMSoftClause(MI))
-      break;
-
-    addClauseInst(*MI);
-  }
+  if (ClauseKind != getSoftClauseKind(*MEM))
+    return 0;
 
   if (ClauseDefs.none())
     return 0;
@@ -1171,11 +1186,11 @@ int GCNHazardRecognizer::checkSoftClauseHazards(MachineInstr *MEM) const {
   if (MEM->mayStore())
     return 1;
 
-  addClauseInst(*MEM);
-
   // If the set of defs and uses intersect then we cannot add this instruction
   // to the clause, so we have a hazard.
-  return ClauseDefs.anyCommon(ClauseUses) ? 1 : 0;
+  BitVector Defs = ClauseDefs, Uses = ClauseUses;
+  addRegsToSet(TRI, MEM->operands(), Defs, Uses);
+  return Defs.anyCommon(Uses) ? 1 : 0;
 }
 
 int GCNHazardRecognizer::checkSMRDHazards(MachineInstr *SMRD) const {
diff --git a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
index 422da557eb47d..168d563a68723 100644
--- a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
+++ b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
@@ -151,18 +151,26 @@ class GCNHazardRecognizer final : public ScheduleHazardRecognizer {
   /// Scheduler-mode part of Reset().
   void schedulerReset();
 
+  /// Tracked in emission order because a clause can exceed getMaxLookAhead()
+  /// and so cannot be reconstructed from EmittedInstrs.
+  enum class SoftClauseKind { None, SMEM, VMEM };
+  SoftClauseKind ClauseKind = SoftClauseKind::None;
+
   /// RegUnits of uses in the current soft memory clause.
-  mutable BitVector ClauseUses;
+  BitVector ClauseUses;
 
   /// RegUnits of defs in the current soft memory clause.
-  mutable BitVector ClauseDefs;
+  BitVector ClauseDefs;
 
-  void resetClause() const {
+  void resetClause() {
+    ClauseKind = SoftClauseKind::None;
     ClauseUses.reset();
     ClauseDefs.reset();
   }
 
-  void addClauseInst(const MachineInstr &MI) const;
+  static SoftClauseKind getSoftClauseKind(const MachineInstr &MI);
+
+  void updateSoftClause(const MachineInstr &MI);
 
   /// \returns the number of wait states before another MFMA instruction can be
   /// issued after \p MI.
diff --git a/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll b/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll
index 9ccbeee37a03a..a21ff31f8c82f 100644
--- a/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll
+++ b/llvm/test/CodeGen/AMDGPU/GlobalISel/vni8-across-blocks.ll
@@ -282,6 +282,7 @@ define amdgpu_kernel void @v256i8_liveout(ptr addrspace(1) %src1, ptr addrspace(
 ; GFX906-NEXT:    global_load_dwordx4 v[53:56], v4, s[0:1] offset:208
 ; GFX906-NEXT:    global_load_dwordx4 v[57:60], v4, s[0:1] offset:224
 ; GFX906-NEXT:    global_load_dwordx4 v[0:3], v4, s[0:1] offset:240
+; GFX906-NEXT:    s_nop 0
 ; GFX906-NEXT:    global_load_dwordx4 v[5:8], v4, s[0:1] offset:16
 ; GFX906-NEXT:    global_load_dwordx4 v[9:12], v4, s[0:1] offset:32
 ; GFX906-NEXT:    global_load_dwordx4 v[13:16], v4, s[0:1] offset:48
@@ -309,6 +310,7 @@ define amdgpu_kernel void @v256i8_liveout(ptr addrspace(1) %src1, ptr addrspace(
 ; GFX906-NEXT:    global_load_dwordx4 v[49:52], v4, s[2:3] offset:192
 ; GFX906-NEXT:    global_load_dwordx4 v[53:56], v4, s[2:3] offset:208
 ; GFX906-NEXT:    global_load_dwordx4 v[57:60], v4, s[2:3] offset:224
+; GFX906-NEXT:    s_nop 0
 ; GFX906-NEXT:    global_load_dwordx4 v[0:3], v4, s[2:3] offset:240
 ; GFX906-NEXT:  .LBB6_2: ; %bb.2
 ; GFX906-NEXT:    s_or_b64 exec, exec, s[0:1]
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
index 7be6ad4575d69..90bf515d620ea 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
@@ -142895,6 +142895,7 @@ define <64 x bfloat> @bitcast_v128i8_to_v64bf16(<128 x i8> %a, i32 %b) #0 {
 ; GFX9-NEXT:    buffer_load_ushort v60, off, s[0:3], s32 offset:4
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:124
 ; GFX9-NEXT:    buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:372
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -148279,6 +148280,7 @@ define inreg <64 x bfloat> @bitcast_v128i8_to_v64bf16_scalar(<128 x i8> inreg %a
 ; GFX9-NEXT:    buffer_load_ushort v13, off, s[0:3], s32 offset:248
 ; GFX9-NEXT:    buffer_load_ushort v34, off, s[0:3], s32 offset:244
 ; GFX9-NEXT:    buffer_load_ushort v49, off, s[0:3], s32 offset:240
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:236
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:460 ; 4-byte Folded Spill
@@ -155072,6 +155074,7 @@ define <128 x i8> @bitcast_v64bf16_to_v128i8(<64 x bfloat> %a, i32 %b) #0 {
 ; GFX9-NEXT:    buffer_load_dword v56, off, s[0:3], s32 offset:40 ; 4-byte Folded Reload
 ; GFX9-NEXT:    buffer_load_dword v47, off, s[0:3], s32 offset:44 ; 4-byte Folded Reload
 ; GFX9-NEXT:    buffer_load_dword v46, off, s[0:3], s32 offset:48 ; 4-byte Folded Reload
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_dword v45, off, s[0:3], s32 offset:52 ; 4-byte Folded Reload
 ; GFX9-NEXT:    buffer_load_dword v44, off, s[0:3], s32 offset:56 ; 4-byte Folded Reload
 ; GFX9-NEXT:    buffer_load_dword v43, off, s[0:3], s32 offset:60 ; 4-byte Folded Reload
@@ -169184,6 +169187,7 @@ define <64 x half> @bitcast_v128i8_to_v64f16(<128 x i8> %a, i32 %b) #0 {
 ; GFX9-NEXT:    buffer_load_ushort v60, off, s[0:3], s32 offset:4
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:124
 ; GFX9-NEXT:    buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:372
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -174573,6 +174577,7 @@ define inreg <64 x half> @bitcast_v128i8_to_v64f16_scalar(<128 x i8> inreg %a, i
 ; GFX9-NEXT:    buffer_load_ushort v13, off, s[0:3], s32 offset:248
 ; GFX9-NEXT:    buffer_load_ushort v34, off, s[0:3], s32 offset:244
 ; GFX9-NEXT:    buffer_load_ushort v49, off, s[0:3], s32 offset:240
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:236
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:460 ; 4-byte Folded Spill
@@ -189889,6 +189894,7 @@ define <64 x i16> @bitcast_v128i8_to_v64i16(<128 x i8> %a, i32 %b) #0 {
 ; GFX9-NEXT:    buffer_load_ushort v60, off, s[0:3], s32 offset:4
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:124
 ; GFX9-NEXT:    buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:372
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -195278,6 +195284,7 @@ define inreg <64 x i16> @bitcast_v128i8_to_v64i16_scalar(<128 x i8> inreg %a, i3
 ; GFX9-NEXT:    buffer_load_ushort v13, off, s[0:3], s32 offset:248
 ; GFX9-NEXT:    buffer_load_ushort v34, off, s[0:3], s32 offset:244
 ; GFX9-NEXT:    buffer_load_ushort v49, off, s[0:3], s32 offset:240
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:236
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:460 ; 4-byte Folded Spill
diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
index 280a0f17a4bc8..e29b023703a59 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.512bit.ll
@@ -69258,6 +69258,7 @@ define <32 x i16> @bitcast_v64i8_to_v32i16(<64 x i8> %a, i32 %b) #0 {
 ; GFX9-NEXT:    buffer_store_dword v14, off, s[0:3], s32 offset:252 ; 4-byte Folded Spill
 ; GFX9-NEXT:    buffer_store_dword v13, off, s[0:3], s32 offset:256 ; 4-byte Folded Spill
 ; GFX9-NEXT:    buffer_load_ushort v48, off, s[0:3], s32 offset:36
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v60, off, s[0:3], s32 offset:28
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:24
 ; GFX9-NEXT:    buffer_load_ushort v63, off, s[0:3], s32 offset:20
@@ -81995,6 +81996,7 @@ define <32 x half> @bitcast_v64i8_to_v32f16(<64 x i8> %a, i32 %b) #0 {
 ; GFX9-NEXT:    buffer_store_dword v14, off, s[0:3], s32 offset:252 ; 4-byte Folded Spill
 ; GFX9-NEXT:    buffer_store_dword v13, off, s[0:3], s32 offset:256 ; 4-byte Folded Spill
 ; GFX9-NEXT:    buffer_load_ushort v48, off, s[0:3], s32 offset:36
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v60, off, s[0:3], s32 offset:28
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:24
 ; GFX9-NEXT:    buffer_load_ushort v63, off, s[0:3], s32 offset:20
@@ -92665,6 +92667,7 @@ define <32 x bfloat> @bitcast_v64i8_to_v32bf16(<64 x i8> %a, i32 %b) #0 {
 ; GFX9-NEXT:    buffer_store_dword v14, off, s[0:3], s32 offset:252 ; 4-byte Folded Spill
 ; GFX9-NEXT:    buffer_store_dword v13, off, s[0:3], s32 offset:256 ; 4-byte Folded Spill
 ; GFX9-NEXT:    buffer_load_ushort v48, off, s[0:3], s32 offset:36
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v60, off, s[0:3], s32 offset:28
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:24
 ; GFX9-NEXT:    buffer_load_ushort v63, off, s[0:3], s32 offset:20
diff --git a/llvm/test/CodeGen/AMDGPU/break-smem-soft-clauses.mir b/llvm/test/CodeGen/AMDGPU/break-smem-soft-clauses.mir
index 8c5ef9cd5525b..b9e4a48aec961 100644
--- a/llvm/test/CodeGen/AMDGPU/break-smem-soft-clauses.mir
+++ b/llvm/test/CodeGen/AMDGPU/break-smem-soft-clauses.mir
@@ -351,3 +351,132 @@ body: |
     $sgpr12_sgpr13 = S_LOAD_DWORDX2_IMM $sgpr6_sgpr7, 0, 0
     S_ENDPGM 0
 ...
+---
+# WAR at clause distance 6. This is beyond the scheduler lookahead window
+# (MaxLookAhead = 5), so the clause must be tracked statefully rather than
+# reconstructed from the emitted-instruction window.
+name: break_smem_clause_war_distance_6
+
+body: |
+  bb.0:
+    ; GCN-LABEL: name: break_smem_clause_war_distance_6
+    ; GCN: $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+    ; GCN-NEXT: $sgpr1 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+    ; GCN-NEXT: $sgpr2 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+    ; GCN-NEXT: $sgpr3 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+    ; GCN-NEXT: $sgpr4 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+    ; GCN-NEXT: $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+    ; XNACK-NEXT: S_NOP 0
+    ; GCN-NEXT: $sgpr10 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+    ; GCN-NEXT: S_ENDPGM 0
+    $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+    $sgpr1 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+    $sgpr2 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+    $sgpr3 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+    $sgpr4 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+    $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+    $sgpr10 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+    S_ENDPGM 0
+...
+---
+# Same as above without the conflict: a long clause on its own must not get a
+# nop.
+name: smem_clause_distance_6_no_conflict
+
+body: |
+  bb.0:
+    ; GCN-LABEL: name: smem_clause_distance_6_no_conflict
+    ; GCN: $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+    ; GCN-NEXT: $sgpr1 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+    ; GCN-NEXT: $sgpr2 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+    ; GCN-NEXT: $sgpr3 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+    ; GCN-NEXT: $sgpr4 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+    ; GCN-NEXT: $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+    ; GCN-NEXT: $sgpr6 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+    ; GCN-NEXT: S_ENDPGM 0
+    $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+    $sgpr1 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+    $sgpr2 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+    $sgpr3 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+    $sgpr4 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+    $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+    $sgpr6 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+    S_ENDPGM 0
+...
+---
+# KILL is a meta instruction with no encoding, so it does not break the
+# hardware clause and the WAR across it must still be detected.
+name: kill_does_not_break_smem_clause
+
+body: |
+  bb.0:
+    ; GCN-LABEL: name: kill_does_not_break_smem_clause
+    ; GCN: $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+    ; GCN-NEXT: KILL $sgpr8_sgpr9
+    ; XNACK-NEXT: S_NOP 0
+    ; GCN-NEXT: $sgpr10 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+    ; GCN-NEXT: S_ENDPGM 0
+    $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+    KILL $sgpr8_sgpr9
+    $sgpr10 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+    S_ENDPGM 0
+...
+---
+# A KILL's operands are not part of the clause: overwriting them is fine.
+name: kill_operand_does_not_join_smem_clause
+
+body: |
+  bb.0:
+    ; GCN-LABEL: name: kill_operand_does_not_join_smem_clause
+    ; GCN: $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+    ; GCN-NEXT: KILL $sgpr8_sgpr9
+    ; GCN-NEXT: $sgpr8 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+    ; GCN-NEXT: S_ENDPGM 0
+    $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+    KILL $sgpr8_sgpr9
+    $sgpr8 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+    S_ENDPGM 0
+...
+---
+# The shape SIFormMemoryClauses leaves at its 15-instruction limit: a run of
+# loads, the KILLs that ended the first bundle, then more loads. The KILLs do
+# not break the hardware clause, and load 17 overwrites load 1's address.
+name: break_smem_clause_kill_boundary_17_loads
+
+body: |
+  bb.0:
+    ; GCN-LABEL: name: break_smem_clause_kill_boundary_17_loads
+    ; GCN: $sgpr40 = S_LOAD_DWORD_IMM $sgpr2_sgpr3, 0, 0
+    ; GCN: $sgpr51 = S_LOAD_DWORD_IMM $sgpr24_sgpr25, 0, 0
+    ; GCN-NEXT: KILL killed renamable $sgpr2_sgpr3
+    ; GCN-NEXT: KILL killed renamable $sgpr4_sgpr5
+    ; GCN-NEXT: KILL killed renamable $sgpr6_sgpr7
+    ; GCN-NEXT: $sgpr52 = S_LOAD_DWORD_IMM $sgpr26_sgpr27, 0, 0
+    ; GCN-NEXT: $sgpr53 = S_LOAD_DWORD_IMM $sgpr28_sgpr29, 0, 0
+    ; GCN-NEXT: $sgpr54 = S_LOAD_DWORD_IMM $sgpr30_sgpr31, 0, 0
+    ; GCN-NEXT: $sgpr55 = S_LOAD_DWORD_IMM $sgpr32_sgpr33, 0, 0
+    ; XNACK-NEXT: S_NOP 0
+    ; GCN-NEXT: $sgpr2 = S_LOAD_DWORD_IMM $sgpr34_sgpr35, 0, 0
+    ; GCN-NEXT: S_ENDPGM 0
+    $sgpr40 = S_LOAD_DWORD_IMM $sgpr2_sgpr3, 0, 0
+    $sgpr41 = S_LOAD_DWORD_IMM $sgpr4_sgpr5, 0, 0
+    $sgpr42 = S_LOAD_DWORD_IMM $sgpr6_sgpr7, 0, 0
+    $sgpr43 = S_LOAD_DWORD_IMM $sgpr8_sgpr9, 0, 0
+    $sgpr44 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+    $sgpr45 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+    $sgpr46 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+    $sgpr47 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+    $sgpr48 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+    $sgpr49 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+    $sgpr50 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+    $sgpr51 = S_LOAD_DWORD_IMM $sgpr24_sgpr25, 0, 0
+    KILL killed renamable $sgpr2_sgpr3
+    KILL killed renamable $sgpr4_sgpr5
+    KILL killed renamable $sgpr6_sgpr7
+    $sgpr52 = S_LOAD_DWORD_IMM $sgpr26_sgpr27, 0, 0
+    $sgpr53 = S_LOAD_DWORD_IMM $sgpr28_sgpr29, 0, 0
+    $sgpr54 = S_LOAD_DWORD_IMM $sgpr30_sgpr31, 0, 0
+    $sgpr55 = S_LOAD_DWORD_IMM $sgpr32_sgpr33, 0, 0
+    $sgpr2 = S_LOAD_DWORD_IMM $sgpr34_sgpr35, 0, 0
+    S_ENDPGM 0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/break-vmem-soft-clauses.mir b/llvm/test/CodeGen/AMDGPU/break-vmem-soft-clauses.mir
index 55460fd0d6d20..9c6a0a5554424 100644
--- a/llvm/test/CodeGen/AMDGPU/break-vmem-soft-clauses.mir
+++ b/llvm/test/CodeGen/AMDGPU/break-vmem-soft-clauses.mir
@@ -577,3 +577,29 @@ body: |
     $vgpr11 = FLAT_LOAD_DWORD $vgpr4_vgpr5, 0, 0, implicit $exec, implicit $flat_scr
     S_ENDPGM 0
 ...
+---
+# WAR at clause distance 6, beyond the scheduler lookahead window
+# (MaxLookAhead = 5).
+name: break_vmem_clause_war_distance_6
+
+body: |
+  bb.0:
+    ; GCN-LABEL: name: break_vmem_clause_war_distance_6
+    ; GCN: $vgpr0 = FLAT_LOAD_DWORD $vgpr2_vgpr3, 0, 0, implicit $exec, implicit $flat_scr
+    ; GCN-NEXT: $vgpr1 = FLAT_LOAD_DWORD $vgpr4_vgpr5, 0, 0, implicit $exec, implicit $flat_scr
+    ; GCN-NEXT: $vgpr10 = FLAT_LOAD_DWORD $vgpr6_vgpr7, 0, 0, implicit $exec, implicit $flat_scr
+    ; GCN-NEXT: $vgpr11 = FLAT_LOAD_DWORD $vgpr8_vgpr9, 0, 0, implicit $exec, implicit $flat_scr
+    ; GCN-NEXT: $vgpr12 = FLAT_LOAD_DWORD $vgpr14_vgpr15, 0, 0, implicit $exec, implicit $flat_scr
+    ; GCN-NEXT: $vgpr13 = FLAT_LOAD_DWORD $vgpr16_vgpr17, 0, 0, implicit $exec, implicit $flat_scr
+    ; XNACK-NEXT: S_NOP 0
+    ; GCN-NEXT: $vgpr2 = FLAT_LOAD_DWORD $vgpr18_vgpr19, 0, 0, implicit $exec, implicit $flat_scr
+    ; GCN-NEXT: S_ENDPGM 0
+    $vgpr0 = FLAT_LOAD_DWORD $vgpr2_vgpr3, 0, 0, implicit $exec, implicit $flat_scr
+    $vgpr1 = FLAT_LOAD_DWORD $vgpr4_vgpr5, 0, 0, implicit $exec, implicit $flat_scr
+    $vgpr10 = FLAT_LOAD_DWORD $vgpr6_vgpr7, 0, 0, implicit $exec, implicit $flat_scr
+    $vgpr11 = FLAT_LOAD_DWORD $vgpr8_vgpr9, 0, 0, implicit $exec, implicit $flat_scr
+    $vgpr12 = FLAT_LOAD_DWORD $vgpr14_vgpr15, 0, 0, implicit $exec, implicit $flat_scr
+    $vgpr13 = FLAT_LOAD_DWORD $vgpr16_vgpr17, 0, 0, implicit $exec, implicit $flat_scr
+    $vgpr2 = FLAT_LOAD_DWORD $vgpr18_vgpr19, 0, 0, implicit $exec, implicit $flat_scr
+    S_ENDPGM 0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/gfx-callable-return-types.ll b/llvm/test/CodeGen/AMDGPU/gfx-callable-return-types.ll
index f805b11d35db5..782df520b817d 100644
--- a/llvm/test/CodeGen/AMDGPU/gfx-callable-return-types.ll
+++ b/llvm/test/CodeGen/AMDGPU/gfx-callable-return-types.ll
@@ -2900,6 +2900,7 @@ define amdgpu_gfx void @call_72xi32() #1 {
 ; GFX9-NEXT:    buffer_store_dword v7, off, s[0:3], s32 offset:156
 ; GFX9-NEXT:    buffer_store_dword v8, off, s[0:3], s32 offset:160
 ; GFX9-NEXT:    buffer_load_dword v2, off, s[0:3], s33 offset:1536 ; 4-byte Folded Reload
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_dword v3, off, s[0:3], s33 offset:1540 ; 4-byte Folded Reload
 ; GFX9-NEXT:    buffer_load_dword v4, off, s[0:3], s33 offset:1544 ; 4-byte Folded Reload
 ; GFX9-NEXT:    buffer_load_dword v5, off, s[0:3], s33 offset:1548 ; 4-byte Folded Reload
diff --git a/llvm/test/CodeGen/AMDGPU/hazard-in-bundle.mir b/llvm/test/CodeGen/AMDGPU/hazard-in-bundle.mir
index 3cc5d455d36d8..0a3477f625db4 100644
--- a/llvm/test/CodeGen/AMDGPU/hazard-in-bundle.mir
+++ b/llvm/test/CodeGen/AMDGPU/hazard-in-bundle.mir
@@ -82,3 +82,25 @@ body: |
     }
     S_ENDPGM 0
 ...
+
+# GCN-LABEL:    name: break_smem_clause_beyond_max_look_ahead_in_bundle
+# GCN:          $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+# XNACK-NEXT:   S_NOP
+# NOXNACK-NOT:  S_NOP
+# GCN-NEXT:     $sgpr10 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+# GCN:          }
+---
+name: break_smem_clause_beyond_max_look_ahead_in_bundle
+body: |
+  bb.0:
+    BUNDLE implicit-def $sgpr0, implicit-def $sgpr1, implicit-def $sgpr2, implicit-def $sgpr3, implicit-def $sgpr4, implicit-def $sgpr5, implicit-def $sgpr10, implicit $sgpr10_sgpr11, implicit $sgpr12_sgpr13, implicit $sgpr14_sgpr15, implicit $sgpr16_sgpr17, implicit $sgpr18_sgpr19, implicit $sgpr20_sgpr21, implicit $sgpr22_sgpr23 {
+      $sgpr0 = S_LOAD_DWORD_IMM $sgpr10_sgpr11, 0, 0
+      $sgpr1 = S_LOAD_DWORD_IMM $sgpr12_sgpr13, 0, 0
+      $sgpr2 = S_LOAD_DWORD_IMM $sgpr14_sgpr15, 0, 0
+      $sgpr3 = S_LOAD_DWORD_IMM $sgpr16_sgpr17, 0, 0
+      $sgpr4 = S_LOAD_DWORD_IMM $sgpr18_sgpr19, 0, 0
+      $sgpr5 = S_LOAD_DWORD_IMM $sgpr20_sgpr21, 0, 0
+      $sgpr10 = S_LOAD_DWORD_IMM $sgpr22_sgpr23, 0, 0
+    }
+    S_ENDPGM 0
+...
diff --git a/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll b/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll
index ab946d8667f8c..87692260e90e2 100644
--- a/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll
+++ b/llvm/test/CodeGen/AMDGPU/splitkit-getsubrangeformask-phi-extend.ll
@@ -299,6 +299,7 @@ define void @f(ptr %p, <16 x i1> %m, <16 x i63> %pt, <16 x i1> %sc,
 ; CHECK-NEXT:    buffer_load_dword a9, off, s[0:3], s32 offset:456
 ; CHECK-NEXT:    buffer_load_dword v12, off, s[0:3], s32 offset:8
 ; CHECK-NEXT:    buffer_load_dword v10, off, s[0:3], s32 offset:4
+; CHECK-NEXT:    s_nop 0
 ; CHECK-NEXT:    buffer_load_dword v6, off, s[0:3], s32
 ; CHECK-NEXT:    buffer_load_dword v24, off, s[0:3], s32 offset:708
 ; CHECK-NEXT:    buffer_load_dword v56, off, s[0:3], s32 offset:704

>From a3bd0b2c9e937e268055cdbb2829db4ee1a1fdee Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Fri, 11 Sep 2026 11:01:14 +0200
Subject: [PATCH 2/3] upd

---
 llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll | 13 +++++++++++++
 1 file changed, 13 insertions(+)

diff --git a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
index 62ad9928797e2..b3e2f6ef88fab 100644
--- a/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
+++ b/llvm/test/CodeGen/AMDGPU/amdgcn.bitcast.1024bit.ll
@@ -20107,6 +20107,7 @@ define inreg <32 x i32> @bitcast_v128i8_to_v32i32_scalar(<128 x i8> inreg %a, i3
 ; GFX9-NEXT:    buffer_load_ushort v25, off, s[0:3], s32 offset:140
 ; GFX9-NEXT:    buffer_load_ushort v24, off, s[0:3], s32 offset:136
 ; GFX9-NEXT:    buffer_load_ushort v23, off, s[0:3], s32 offset:132
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:128
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:124
 ; GFX9-NEXT:    buffer_load_ushort v18, off, s[0:3], s32 offset:120
@@ -56225,6 +56226,7 @@ define inreg <32 x float> @bitcast_v128i8_to_v32f32_scalar(<128 x i8> inreg %a,
 ; GFX9-NEXT:    buffer_load_ushort v25, off, s[0:3], s32 offset:140
 ; GFX9-NEXT:    buffer_load_ushort v24, off, s[0:3], s32 offset:136
 ; GFX9-NEXT:    buffer_load_ushort v23, off, s[0:3], s32 offset:132
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:128
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:124
 ; GFX9-NEXT:    buffer_load_ushort v18, off, s[0:3], s32 offset:120
@@ -90562,6 +90564,7 @@ define inreg <16 x i64> @bitcast_v128i8_to_v16i64_scalar(<128 x i8> inreg %a, i3
 ; GFX9-NEXT:    buffer_load_ushort v25, off, s[0:3], s32 offset:140
 ; GFX9-NEXT:    buffer_load_ushort v24, off, s[0:3], s32 offset:136
 ; GFX9-NEXT:    buffer_load_ushort v23, off, s[0:3], s32 offset:132
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:128
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:124
 ; GFX9-NEXT:    buffer_load_ushort v18, off, s[0:3], s32 offset:120
@@ -123799,6 +123802,7 @@ define inreg <16 x double> @bitcast_v128i8_to_v16f64_scalar(<128 x i8> inreg %a,
 ; GFX9-NEXT:    buffer_load_ushort v25, off, s[0:3], s32 offset:140
 ; GFX9-NEXT:    buffer_load_ushort v24, off, s[0:3], s32 offset:136
 ; GFX9-NEXT:    buffer_load_ushort v23, off, s[0:3], s32 offset:132
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:128
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:124
 ; GFX9-NEXT:    buffer_load_ushort v18, off, s[0:3], s32 offset:120
@@ -143855,6 +143859,7 @@ define <64 x bfloat> @bitcast_v128i8_to_v64bf16(<128 x i8> %a, i32 %b) #0 {
 ; GFX9-NEXT:    buffer_load_ushort v60, off, s[0:3], s32 offset:4
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:124
 ; GFX9-NEXT:    buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:372
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -149251,6 +149256,7 @@ define inreg <64 x bfloat> @bitcast_v128i8_to_v64bf16_scalar(<128 x i8> inreg %a
 ; GFX9-NEXT:    buffer_load_ushort v6, off, s[0:3], s32 offset:272
 ; GFX9-NEXT:    buffer_load_ushort v8, off, s[0:3], s32 offset:268
 ; GFX9-NEXT:    buffer_load_ushort v41, off, s[0:3], s32 offset:264
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v2, off, s[0:3], s32 offset:260
 ; GFX9-NEXT:    buffer_load_ushort v24, off, s[0:3], s32 offset:256
 ; GFX9-NEXT:    buffer_load_ushort v9, off, s[0:3], s32 offset:252
@@ -149348,6 +149354,7 @@ define inreg <64 x bfloat> @bitcast_v128i8_to_v64bf16_scalar(<128 x i8> inreg %a
 ; GFX9-NEXT:    buffer_load_ushort v30, off, s[0:3], s32 offset:60
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:56
 ; GFX9-NEXT:    buffer_load_ushort v52, off, s[0:3], s32 offset:52
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v14, off, s[0:3], s32 offset:48
 ; GFX9-NEXT:    buffer_load_ushort v29, off, s[0:3], s32 offset:44
 ; GFX9-NEXT:    buffer_load_ushort v39, off, s[0:3], s32 offset:40
@@ -170343,6 +170350,7 @@ define <64 x half> @bitcast_v128i8_to_v64f16(<128 x i8> %a, i32 %b) #0 {
 ; GFX9-NEXT:    buffer_load_ushort v60, off, s[0:3], s32 offset:4
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:124
 ; GFX9-NEXT:    buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:372
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -175739,6 +175747,7 @@ define inreg <64 x half> @bitcast_v128i8_to_v64f16_scalar(<128 x i8> inreg %a, i
 ; GFX9-NEXT:    buffer_load_ushort v6, off, s[0:3], s32 offset:272
 ; GFX9-NEXT:    buffer_load_ushort v8, off, s[0:3], s32 offset:268
 ; GFX9-NEXT:    buffer_load_ushort v41, off, s[0:3], s32 offset:264
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v2, off, s[0:3], s32 offset:260
 ; GFX9-NEXT:    buffer_load_ushort v24, off, s[0:3], s32 offset:256
 ; GFX9-NEXT:    buffer_load_ushort v9, off, s[0:3], s32 offset:252
@@ -175836,6 +175845,7 @@ define inreg <64 x half> @bitcast_v128i8_to_v64f16_scalar(<128 x i8> inreg %a, i
 ; GFX9-NEXT:    buffer_load_ushort v30, off, s[0:3], s32 offset:60
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:56
 ; GFX9-NEXT:    buffer_load_ushort v52, off, s[0:3], s32 offset:52
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v14, off, s[0:3], s32 offset:48
 ; GFX9-NEXT:    buffer_load_ushort v29, off, s[0:3], s32 offset:44
 ; GFX9-NEXT:    buffer_load_ushort v39, off, s[0:3], s32 offset:40
@@ -191200,6 +191210,7 @@ define <64 x i16> @bitcast_v128i8_to_v64i16(<128 x i8> %a, i32 %b) #0 {
 ; GFX9-NEXT:    buffer_load_ushort v60, off, s[0:3], s32 offset:4
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:124
 ; GFX9-NEXT:    buffer_load_ushort v35, off, s[0:3], s32 offset:120
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v0, off, s[0:3], s32 offset:372
 ; GFX9-NEXT:    s_waitcnt vmcnt(0)
 ; GFX9-NEXT:    buffer_store_dword v0, off, s[0:3], s32 offset:744 ; 4-byte Folded Spill
@@ -196596,6 +196607,7 @@ define inreg <64 x i16> @bitcast_v128i8_to_v64i16_scalar(<128 x i8> inreg %a, i3
 ; GFX9-NEXT:    buffer_load_ushort v6, off, s[0:3], s32 offset:272
 ; GFX9-NEXT:    buffer_load_ushort v8, off, s[0:3], s32 offset:268
 ; GFX9-NEXT:    buffer_load_ushort v41, off, s[0:3], s32 offset:264
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v2, off, s[0:3], s32 offset:260
 ; GFX9-NEXT:    buffer_load_ushort v24, off, s[0:3], s32 offset:256
 ; GFX9-NEXT:    buffer_load_ushort v9, off, s[0:3], s32 offset:252
@@ -196693,6 +196705,7 @@ define inreg <64 x i16> @bitcast_v128i8_to_v64i16_scalar(<128 x i8> inreg %a, i3
 ; GFX9-NEXT:    buffer_load_ushort v30, off, s[0:3], s32 offset:60
 ; GFX9-NEXT:    buffer_load_ushort v62, off, s[0:3], s32 offset:56
 ; GFX9-NEXT:    buffer_load_ushort v52, off, s[0:3], s32 offset:52
+; GFX9-NEXT:    s_nop 0
 ; GFX9-NEXT:    buffer_load_ushort v14, off, s[0:3], s32 offset:48
 ; GFX9-NEXT:    buffer_load_ushort v29, off, s[0:3], s32 offset:44
 ; GFX9-NEXT:    buffer_load_ushort v39, off, s[0:3], s32 offset:40

>From 136fa40d20b4fd40d5ebc1df2a9c3bb3dddd5125 Mon Sep 17 00:00:00 2001
From: Arseniy Obolenskiy <arseniy.obolenskiy at amd.com>
Date: Fri, 11 Sep 2026 16:23:57 +0200
Subject: [PATCH 3/3] comnments

---
 .../lib/Target/AMDGPU/GCNHazardRecognizer.cpp | 45 ++++++++++++++-----
 llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h  | 15 ++++---
 2 files changed, 42 insertions(+), 18 deletions(-)

diff --git a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
index 1edd04293a5db..083fe6ac109bc 100644
--- a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
+++ b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.cpp
@@ -679,12 +679,12 @@ void GCNHazardRecognizer::processBundle() {
     unsigned WaitStates = PreEmitNoopsCommon(CurrCycleInstr);
 
     if (isHazardRecognizerMode()) {
+      // fixHazards can reset CurrCycleInstr to null, so use MI from here on.
       fixHazards(CurrCycleInstr);
 
-      insertNoopsInBundle(CurrCycleInstr, TII, WaitStates);
+      insertNoopsInBundle(&*MI, TII, WaitStates);
     }
 
-    // Use MI, not CurrCycleInstr, which fixHazards may have reset to null.
     if (WaitStates)
       resetClause();
     updateSoftClause(*MI);
@@ -695,7 +695,7 @@ void GCNHazardRecognizer::processBundle() {
     for (unsigned i = 0, e = std::min(WaitStates, MaxLookAhead - 1); i < e; ++i)
       EmittedInstrs.push_front(nullptr);
 
-    EmittedInstrs.push_front(CurrCycleInstr);
+    EmittedInstrs.push_front(&*MI);
     EmittedInstrs.resize(MaxLookAhead);
   }
   CurrCycleInstr = nullptr;
@@ -825,10 +825,10 @@ void GCNHazardRecognizer::AdvanceCycle() {
   // When the scheduler detects a stall, it will call AdvanceCycle() without
   // emitting any instructions.
   if (!CurrCycleInstr) {
+    assert(isSchedulerMode() && "stall cycles only occur in scheduler mode");
     EmittedInstrs.push_front(nullptr);
-    // Only reached in scheduler mode, where clause hazards are a heuristic
-    // and stalling cannot break one; reset here or it stalls forever.
-    assert(isSchedulerMode());
+    // A stall does not really break a clause, but model it as one or the
+    // scheduler stalls on the same hazard forever.
     resetClause();
 
     if (HasPendingWMMACoexecHazard)
@@ -1138,6 +1138,13 @@ static void addRegsToSet(const SIRegisterInfo &TRI,
   }
 }
 
+static bool anyRegUnitSet(const SIRegisterInfo &TRI, const BitVector &Units,
+                          MCRegister Reg) {
+  return any_of(TRI.regunits(Reg), [&Units](MCRegUnit Unit) {
+    return Units.test(static_cast<unsigned>(Unit));
+  });
+}
+
 GCNHazardRecognizer::SoftClauseKind
 GCNHazardRecognizer::getSoftClauseKind(const MachineInstr &MI) {
   if (SIInstrInfo::isSMRD(MI))
@@ -1148,10 +1155,11 @@ GCNHazardRecognizer::getSoftClauseKind(const MachineInstr &MI) {
 }
 
 void GCNHazardRecognizer::updateSoftClause(const MachineInstr &MI) {
-  if (!ST.isXNACKEnabled())
+  // checkSoftClauseHazards is only reached when hasPhysRegs().
+  if (!hasPhysRegs() || !ST.isXNACKEnabled())
     return;
 
-  // Meta instructions have no encoding, so they neither extend nor break a
+  // Meta instructions have no encoding, so they cannot break or extend a
   // clause.
   if (MI.isMetaInstruction())
     return;
@@ -1194,10 +1202,23 @@ int GCNHazardRecognizer::checkSoftClauseHazards(MachineInstr *MEM) const {
     return 1;
 
   // If the set of defs and uses intersect then we cannot add this instruction
-  // to the clause, so we have a hazard.
-  BitVector Defs = ClauseDefs, Uses = ClauseUses;
-  addRegsToSet(TRI, MEM->operands(), Defs, Uses);
-  return Defs.anyCommon(Uses) ? 1 : 0;
+  // to the clause, so we have a hazard. Test MEM in place to avoid copying a
+  // reg-unit-sized BitVector per query.
+  if (ClauseDefs.anyCommon(ClauseUses))
+    return 1;
+
+  for (const MachineOperand &Op : MEM->operands()) {
+    if (!Op.isReg() || !Op.getReg().isPhysical())
+      continue;
+    MCRegister Reg = Op.getReg().asMCReg();
+    if (anyRegUnitSet(TRI, Op.isDef() ? ClauseUses : ClauseDefs, Reg))
+      return 1;
+    // MEM can also conflict with itself.
+    if (Op.isDef() && MEM->readsRegister(Reg, &TRI))
+      return 1;
+  }
+
+  return 0;
 }
 
 int GCNHazardRecognizer::checkSMRDHazards(MachineInstr *SMRD) const {
diff --git a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
index 168d563a68723..4cd417eb3bb2c 100644
--- a/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
+++ b/llvm/lib/Target/AMDGPU/GCNHazardRecognizer.h
@@ -151,18 +151,21 @@ class GCNHazardRecognizer final : public ScheduleHazardRecognizer {
   /// Scheduler-mode part of Reset().
   void schedulerReset();
 
-  /// Tracked in emission order because a clause can exceed getMaxLookAhead()
-  /// and so cannot be reconstructed from EmittedInstrs.
   enum class SoftClauseKind { None, SMEM, VMEM };
-  SoftClauseKind ClauseKind = SoftClauseKind::None;
 
-  /// RegUnits of uses in the current soft memory clause.
+  /// The current soft memory clause, as reg units. It cannot be rebuilt from
+  /// EmittedInstrs: getMaxLookAhead() bounds wait states, not clause length.
+  SoftClauseKind ClauseKind = SoftClauseKind::None;
   BitVector ClauseUses;
-
-  /// RegUnits of defs in the current soft memory clause.
   BitVector ClauseDefs;
 
   void resetClause() {
+    // EmitNoops() calls this once per nop, and clearing is not free.
+    if (ClauseKind == SoftClauseKind::None) {
+      assert(ClauseUses.none() && ClauseDefs.none() &&
+             "no clause kind implies no tracked reg units");
+      return;
+    }
     ClauseKind = SoftClauseKind::None;
     ClauseUses.reset();
     ClauseDefs.reset();



More information about the llvm-commits mailing list