[llvm] [AMDGPU] DAG Mutation to solve load bunching in outer product matrix multiplications (PR #203095)
Axel Sorenson via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 9 18:54:35 PDT 2026
https://github.com/axelcool1234 updated https://github.com/llvm/llvm-project/pull/203095
>From 3b55901c91de3bf2739f4d7d86adc25cc3a06af1 Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Thu, 4 Jun 2026 23:36:28 +0000
Subject: [PATCH 01/20] DAG Mutation to solve ds_load bunching in outer product
matmuls
---
.../lib/Target/AMDGPU/AMDGPUTargetMachine.cpp | 2 +
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 119 ++++++++++++++++++
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.h | 24 ++++
llvm/lib/Target/AMDGPU/CMakeLists.txt | 1 +
4 files changed, 146 insertions(+)
create mode 100644 llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
create mode 100644 llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.h
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
index b078e0835a90e..4d19a8b2cc6a3 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
@@ -39,6 +39,7 @@
#include "AMDGPUTargetTransformInfo.h"
#include "AMDGPUUnifyDivergentExitNodes.h"
#include "AMDGPUWaitSGPRHazards.h"
+#include "AMDGPUWMMASchedule.h"
#include "GCNDPPCombine.h"
#include "GCNIterativeScheduler.h"
#include "GCNNSAReassign.h"
@@ -760,6 +761,7 @@ createGCNMaxOccupancyMachineScheduler(MachineSchedContext *C) {
DAG->addMutation(createAMDGPUExportClusteringDAGMutation());
DAG->addMutation(createAMDGPUBarrierLatencyDAGMutation(C->MF));
DAG->addMutation(createAMDGPUHazardLatencyDAGMutation(C->MF));
+ DAG->addMutation(createAMDGPUWMMAScheduleDAGMutation(C->MF));
return DAG;
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
new file mode 100644
index 0000000000000..e367ad9a22659
--- /dev/null
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -0,0 +1,119 @@
+//===--- AMDGPUWMMASchedule.cpp - AMDGPU WMMA Schedule Adjustment ---------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+//
+/// \file This file contains a DAG scheduling mutation to add additional
+/// edges between ds_load instructions and wmma instructions that
+/// occur a certain amount away from the actual wmma consumer of
+/// said ds_load. This forces the ds_load to properly prefetch
+/// and prevent early bunching of ds_loads that then lead to long
+/// stalls.
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPUWMMASchedule.h"
+#include "GCNSubtarget.h"
+#include "SIInstrInfo.h"
+#include "llvm/CodeGen/ScheduleDAG.h"
+#include "llvm/CodeGen/ScheduleDAGInstrs.h"
+#include "llvm/Support/Debug.h"
+#include <optional>
+#define DEBUG_TYPE "amdgpu-wmma-sched"
+
+using namespace llvm;
+
+namespace {
+
+class WMMASchedule : public ScheduleDAGMutation {
+private:
+ const GCNSubtarget &ST;
+ const SIRegisterInfo &TRI;
+ const MachineRegisterInfo &MRI;
+
+public:
+ WMMASchedule(MachineFunction *MF)
+ : ST(MF->getSubtarget<GCNSubtarget>()), TRI(*ST.getRegisterInfo()),
+ MRI(MF->getRegInfo()) {}
+ void apply(ScheduleDAGInstrs *DAG) override;
+};
+
+void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
+ if (!ST.hasGFX1250Insts()) return;
+ const TargetSchedModel *SM = DAG->getSchedModel();
+ const SIInstrInfo *TII = ST.getInstrInfo();
+ LLVM_DEBUG(dbgs() << "WMMASchedule running, " << DAG->SUnits.size() << "SUnits\n");
+
+ SmallVector<SUnit*> Loads;
+ MapVector<SUnit*, unsigned> Wmmas;
+
+ std::optional<unsigned> LoadLatency = std::nullopt;
+ std::optional<unsigned> WmmaLatency = std::nullopt;
+
+ // Gather all WMMAs and DS_LOADs
+ for(auto &SU : DAG->SUnits) {
+ MachineInstr* MI = SU.getInstr();
+ if(!MI) continue;
+
+ // Gather WMMAs
+ if(TII->isMFMAorWMMA(*MI)) {
+ if(WmmaLatency == std::nullopt) WmmaLatency = SM->computeInstrLatency(MI);
+ Wmmas.insert({&SU, Wmmas.size()});
+ continue;
+ }
+
+ // Gather DS_LOADs
+ if(TII->isDS(*MI) && MI->mayLoad()) {
+ if(LoadLatency == std::nullopt) LoadLatency = SM->computeInstrLatency(MI);
+ Loads.push_back(&SU);
+ }
+ }
+
+ // Calculate how many WMMAs away from consuming WMMA a load must be
+ // before it will certainly ready for consumer
+ unsigned Dist;
+ if(LoadLatency && WmmaLatency) {
+ Dist = std::ceil(static_cast<double>(*LoadLatency) / *WmmaLatency);
+ if (Dist < 1) Dist = 1;
+ } else {
+ return; // Either missing Wmmas or Loads. No point continuing.
+ }
+ LLVM_DEBUG(dbgs() << "Dist " << Dist << "\n");
+
+ // For every load, determine earliest WMMA reliant on it,
+ // and add an anchor.
+ for(SUnit* L : Loads){
+ SUnit* Earliest = nullptr;
+ unsigned EarliestPos = UINT_MAX;
+ for(const SDep& D : L->Succs){
+ if(D.getKind() != SDep::Data) continue;
+ SUnit* S = D.getSUnit();
+ auto *It = Wmmas.find(S);
+ if(It == Wmmas.end()) continue;
+ if(It->second < EarliestPos) {
+ EarliestPos = It->second;
+ Earliest = S;
+ }
+ };
+ LLVM_DEBUG(dbgs() << "load SU" << L->NodeNum << " -> earliest WMMA SU"
+ << (Earliest ? (int)Earliest->NodeNum : -1)
+ << " (pos " << EarliestPos << ")\n");
+ if(Earliest && EarliestPos >= Dist) {
+ LLVM_DEBUG(dbgs() << "Window Created!\n");
+ SUnit* Anchor = Wmmas.begin()[EarliestPos - Dist].first;
+ bool Ok = DAG->addEdge(L, SDep(Anchor, SDep::Artificial));
+ LLVM_DEBUG(dbgs() << " leash SU" << L->NodeNum << " after WMMA SU" << Anchor->NodeNum
+ << (Ok ? "\n" : " (REJECTED: cycle)\n"));
+ };
+ };
+}
+
+} // end namespace
+
+std::unique_ptr<ScheduleDAGMutation>
+llvm::createAMDGPUWMMAScheduleDAGMutation(MachineFunction *MF) {
+ return std::make_unique<WMMASchedule>(MF);
+}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.h b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.h
new file mode 100644
index 0000000000000..781f4b3b1ac86
--- /dev/null
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.h
@@ -0,0 +1,24 @@
+//===- AMDGPUWMMASchedule.h - WMMA Schedule Adjustment ----------*- C++ -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef LLVM_LIB_TARGET_AMDGPU_AMDGPUWMMASCHEDULE_H
+#define LLVM_LIB_TARGET_AMDGPU_AMDGPUWMMASCHEDULE_H
+
+#include "llvm/CodeGen/ScheduleDAGMutation.h"
+#include <memory>
+
+namespace llvm {
+
+class MachineFunction;
+
+std::unique_ptr<ScheduleDAGMutation>
+createAMDGPUWMMAScheduleDAGMutation(MachineFunction *MF);
+
+} // namespace llvm
+
+#endif // LLVM_LIB_TARGET_AMDGPU_AMDGPUWMMASCHEDULE_H
diff --git a/llvm/lib/Target/AMDGPU/CMakeLists.txt b/llvm/lib/Target/AMDGPU/CMakeLists.txt
index ae8f1c0fad5ba..82f129b74076c 100644
--- a/llvm/lib/Target/AMDGPU/CMakeLists.txt
+++ b/llvm/lib/Target/AMDGPU/CMakeLists.txt
@@ -122,6 +122,7 @@ add_llvm_target(AMDGPUCodeGen
AMDGPUTargetTransformInfo.cpp
AMDGPUWaitcntUtils.cpp
AMDGPUWaitSGPRHazards.cpp
+ AMDGPUWMMASchedule.cpp
AMDGPUUnifyDivergentExitNodes.cpp
R600MachineCFGStructurizer.cpp
GCNCreateVOPD.cpp
>From b2209d8b71beffd14c448d1f7b55544e7db498ea Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Wed, 10 Jun 2026 20:23:37 +0000
Subject: [PATCH 02/20] Clang format
---
.../lib/Target/AMDGPU/AMDGPUTargetMachine.cpp | 2 +-
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 69 +++++++++++--------
2 files changed, 40 insertions(+), 31 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
index 4d19a8b2cc6a3..48673fc0d2ad1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
@@ -38,8 +38,8 @@
#include "AMDGPUTargetObjectFile.h"
#include "AMDGPUTargetTransformInfo.h"
#include "AMDGPUUnifyDivergentExitNodes.h"
-#include "AMDGPUWaitSGPRHazards.h"
#include "AMDGPUWMMASchedule.h"
+#include "AMDGPUWaitSGPRHazards.h"
#include "GCNDPPCombine.h"
#include "GCNIterativeScheduler.h"
#include "GCNNSAReassign.h"
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index e367ad9a22659..d87326ba00b49 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -42,42 +42,48 @@ class WMMASchedule : public ScheduleDAGMutation {
};
void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
- if (!ST.hasGFX1250Insts()) return;
+ if (!ST.hasGFX1250Insts())
+ return;
const TargetSchedModel *SM = DAG->getSchedModel();
const SIInstrInfo *TII = ST.getInstrInfo();
- LLVM_DEBUG(dbgs() << "WMMASchedule running, " << DAG->SUnits.size() << "SUnits\n");
+ LLVM_DEBUG(dbgs() << "WMMASchedule running, " << DAG->SUnits.size()
+ << "SUnits\n");
- SmallVector<SUnit*> Loads;
- MapVector<SUnit*, unsigned> Wmmas;
+ SmallVector<SUnit *> Loads;
+ MapVector<SUnit *, unsigned> Wmmas;
std::optional<unsigned> LoadLatency = std::nullopt;
std::optional<unsigned> WmmaLatency = std::nullopt;
-
+
// Gather all WMMAs and DS_LOADs
- for(auto &SU : DAG->SUnits) {
- MachineInstr* MI = SU.getInstr();
- if(!MI) continue;
+ for (auto &SU : DAG->SUnits) {
+ MachineInstr *MI = SU.getInstr();
+ if (!MI)
+ continue;
// Gather WMMAs
- if(TII->isMFMAorWMMA(*MI)) {
- if(WmmaLatency == std::nullopt) WmmaLatency = SM->computeInstrLatency(MI);
+ if (TII->isMFMAorWMMA(*MI)) {
+ if (WmmaLatency == std::nullopt)
+ WmmaLatency = SM->computeInstrLatency(MI);
Wmmas.insert({&SU, Wmmas.size()});
continue;
}
// Gather DS_LOADs
- if(TII->isDS(*MI) && MI->mayLoad()) {
- if(LoadLatency == std::nullopt) LoadLatency = SM->computeInstrLatency(MI);
+ if (TII->isDS(*MI) && MI->mayLoad()) {
+ if (LoadLatency == std::nullopt)
+ LoadLatency = SM->computeInstrLatency(MI);
Loads.push_back(&SU);
}
}
- // Calculate how many WMMAs away from consuming WMMA a load must be
+ // Calculate how many WMMAs away from consuming WMMA a load must be
// before it will certainly ready for consumer
unsigned Dist;
- if(LoadLatency && WmmaLatency) {
- Dist = std::ceil(static_cast<double>(*LoadLatency) / *WmmaLatency);
- if (Dist < 1) Dist = 1;
+ if (LoadLatency && WmmaLatency) {
+ Dist = std::ceil(static_cast<double>(*LoadLatency) / *WmmaLatency);
+ if (Dist < 1)
+ Dist = 1;
} else {
return; // Either missing Wmmas or Loads. No point continuing.
}
@@ -85,28 +91,31 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
// For every load, determine earliest WMMA reliant on it,
// and add an anchor.
- for(SUnit* L : Loads){
- SUnit* Earliest = nullptr;
+ for (SUnit *L : Loads) {
+ SUnit *Earliest = nullptr;
unsigned EarliestPos = UINT_MAX;
- for(const SDep& D : L->Succs){
- if(D.getKind() != SDep::Data) continue;
- SUnit* S = D.getSUnit();
+ for (const SDep &D : L->Succs) {
+ if (D.getKind() != SDep::Data)
+ continue;
+ SUnit *S = D.getSUnit();
auto *It = Wmmas.find(S);
- if(It == Wmmas.end()) continue;
- if(It->second < EarliestPos) {
- EarliestPos = It->second;
+ if (It == Wmmas.end())
+ continue;
+ if (It->second < EarliestPos) {
+ EarliestPos = It->second;
Earliest = S;
}
};
LLVM_DEBUG(dbgs() << "load SU" << L->NodeNum << " -> earliest WMMA SU"
- << (Earliest ? (int)Earliest->NodeNum : -1)
- << " (pos " << EarliestPos << ")\n");
- if(Earliest && EarliestPos >= Dist) {
+ << (Earliest ? (int)Earliest->NodeNum : -1) << " (pos "
+ << EarliestPos << ")\n");
+ if (Earliest && EarliestPos >= Dist) {
LLVM_DEBUG(dbgs() << "Window Created!\n");
- SUnit* Anchor = Wmmas.begin()[EarliestPos - Dist].first;
+ SUnit *Anchor = Wmmas.begin()[EarliestPos - Dist].first;
bool Ok = DAG->addEdge(L, SDep(Anchor, SDep::Artificial));
- LLVM_DEBUG(dbgs() << " leash SU" << L->NodeNum << " after WMMA SU" << Anchor->NodeNum
- << (Ok ? "\n" : " (REJECTED: cycle)\n"));
+ LLVM_DEBUG(dbgs() << " leash SU" << L->NodeNum << " after WMMA SU"
+ << Anchor->NodeNum
+ << (Ok ? "\n" : " (REJECTED: cycle)\n"));
};
};
}
>From acb921155ed503232beee8702b40e9a0f21529ba Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Wed, 10 Jun 2026 23:25:51 +0000
Subject: [PATCH 03/20] Maximum and minimum distance
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 125 +++++++++++-------
1 file changed, 76 insertions(+), 49 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index d87326ba00b49..fba0109a2fcca 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -28,6 +28,14 @@ using namespace llvm;
namespace {
+// A ds_load plus the program order positions (among WMMAs) of its
+// earliest and latest WMMA consumer.
+struct LoadInfo {
+ SUnit *SU;
+ unsigned MinPos = UINT_MAX; // earliest consumer (UINT_MAX means none in this region)
+ unsigned MaxPos = 0; // latest consumer
+};
+
class WMMASchedule : public ScheduleDAGMutation {
private:
const GCNSubtarget &ST;
@@ -46,24 +54,20 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
return;
const TargetSchedModel *SM = DAG->getSchedModel();
const SIInstrInfo *TII = ST.getInstrInfo();
- LLVM_DEBUG(dbgs() << "WMMASchedule running, " << DAG->SUnits.size()
- << "SUnits\n");
-
- SmallVector<SUnit *> Loads;
- MapVector<SUnit *, unsigned> Wmmas;
- std::optional<unsigned> LoadLatency = std::nullopt;
- std::optional<unsigned> WmmaLatency = std::nullopt;
+ // Gather WMMAs (numbered in program order) and ds_loads.
+ MapVector<SUnit *, unsigned> Wmmas; // WMMA SUnit and its program position relative to one another
+ SmallVector<LoadInfo> Loads;
+ std::optional<unsigned> LoadLatency, WmmaLatency;
- // Gather all WMMAs and DS_LOADs
- for (auto &SU : DAG->SUnits) {
+ for (SUnit &SU : DAG->SUnits) {
MachineInstr *MI = SU.getInstr();
if (!MI)
continue;
// Gather WMMAs
if (TII->isMFMAorWMMA(*MI)) {
- if (WmmaLatency == std::nullopt)
+ if (!WmmaLatency)
WmmaLatency = SM->computeInstrLatency(MI);
Wmmas.insert({&SU, Wmmas.size()});
continue;
@@ -71,53 +75,76 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
// Gather DS_LOADs
if (TII->isDS(*MI) && MI->mayLoad()) {
- if (LoadLatency == std::nullopt)
+ if (!LoadLatency)
LoadLatency = SM->computeInstrLatency(MI);
- Loads.push_back(&SU);
+ Loads.push_back({&SU});
}
}
- // Calculate how many WMMAs away from consuming WMMA a load must be
- // before it will certainly ready for consumer
- unsigned Dist;
- if (LoadLatency && WmmaLatency) {
- Dist = std::ceil(static_cast<double>(*LoadLatency) / *WmmaLatency);
- if (Dist < 1)
- Dist = 1;
- } else {
- return; // Either missing Wmmas or Loads. No point continuing.
- }
- LLVM_DEBUG(dbgs() << "Dist " << Dist << "\n");
-
- // For every load, determine earliest WMMA reliant on it,
- // and add an anchor.
- for (SUnit *L : Loads) {
- SUnit *Earliest = nullptr;
- unsigned EarliestPos = UINT_MAX;
- for (const SDep &D : L->Succs) {
+ // The following means the DAG Mutation cannot do anything useful.
+ if (!LoadLatency || !WmmaLatency || Wmmas.empty())
+ return;
+
+ // The number of WMMAs that elapse during one load's latency
+ unsigned Dist = std::ceil(static_cast<double>(*LoadLatency) / *WmmaLatency);
+ if (Dist < 1)
+ Dist = 1;
+
+ // For each load, find the earliest and latest consuming WMMA positions.
+ for (LoadInfo& LI : Loads) {
+ for (const SDep &D : LI.SU->Succs) {
if (D.getKind() != SDep::Data)
continue;
- SUnit *S = D.getSUnit();
- auto *It = Wmmas.find(S);
+ auto *It = Wmmas.find(D.getSUnit());
+ // Check to see if successor is a WMMA
if (It == Wmmas.end())
continue;
- if (It->second < EarliestPos) {
- EarliestPos = It->second;
- Earliest = S;
- }
- };
- LLVM_DEBUG(dbgs() << "load SU" << L->NodeNum << " -> earliest WMMA SU"
- << (Earliest ? (int)Earliest->NodeNum : -1) << " (pos "
- << EarliestPos << ")\n");
- if (Earliest && EarliestPos >= Dist) {
- LLVM_DEBUG(dbgs() << "Window Created!\n");
- SUnit *Anchor = Wmmas.begin()[EarliestPos - Dist].first;
- bool Ok = DAG->addEdge(L, SDep(Anchor, SDep::Artificial));
- LLVM_DEBUG(dbgs() << " leash SU" << L->NodeNum << " after WMMA SU"
- << Anchor->NodeNum
- << (Ok ? "\n" : " (REJECTED: cycle)\n"));
- };
- };
+ LI.MinPos = std::min(LI.MinPos, It->second);
+ LI.MaxPos = std::max(LI.MaxPos, It->second);
+ }
+ }
+
+ // MaxPos in order
+ std::vector<bool> Present(Wmmas.size(), false);
+ for (const LoadInfo &LI : Loads)
+ // Check if there's a consumer for this load in the region
+ if (LI.MinPos != UINT_MAX)
+ Present[LI.MaxPos] = true;
+
+ // Latest position that's <= position Pos at which some load's register frees
+ // A load's register becomes free at its MaxPos, which is its last WMMA consumer.
+ std::vector<std::optional<unsigned>> DeadBy(Wmmas.size(), std::nullopt);
+ std::optional<unsigned> Latest;
+ for (unsigned Pos = 0; Pos < Wmmas.size(); ++Pos) {
+ if (Present[Pos])
+ Latest = Pos;
+ DeadBy[Pos] = Latest;
+ }
+
+ // Create the edges that constrain where the ds_loads can be placed
+ // minimum distance of a load is LI.MinPos - Dist
+ // maximum distance of a load is DeadBy[LI.MinPos - Dist]
+ for (const LoadInfo &LI : Loads) {
+ if (LI.MinPos == UINT_MAX || LI.MinPos < Dist)
+ continue;
+
+ // Minimum distance edge
+ // Load scheduled before this point
+ unsigned MinDist = LI.MinPos - Dist;
+ bool MinSuccess = DAG->addEdge(Wmmas.begin()[MinDist].first, SDep(LI.SU, SDep::Artificial));
+
+ // Maximum distance edge
+ // Load scheduled after this point
+ bool MaxSuccess = true;
+ std::optional<unsigned> MaxDist = DeadBy[MinDist];
+ if (MaxDist && *MaxDist < MinDist)
+ MaxSuccess = DAG->addEdge(LI.SU, SDep(Wmmas.begin()[*MaxDist].first, SDep::Artificial));
+
+ LLVM_DEBUG(dbgs() << "load SU" << LI.SU->NodeNum << " Win=[" << (MaxDist ? *MaxDist : -1) << ","
+ << MinDist << "] MinPos=" << LI.MinPos << " MaxPos=" << LI.MaxPos
+ << (MinSuccess && MaxSuccess ? "\n" : " (edge REJECTED: cycle)\n"));
+ }
+
}
} // end namespace
>From ad5b0a59639b82fa08bf3192d59967313ce890ba Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Thu, 11 Jun 2026 19:34:03 +0000
Subject: [PATCH 04/20] MIR test
---
.../AMDGPU/sched-wmma-ds-load-window.mir | 376 ++++++++++++++++++
1 file changed, 376 insertions(+)
create mode 100644 llvm/test/CodeGen/AMDGPU/sched-wmma-ds-load-window.mir
diff --git a/llvm/test/CodeGen/AMDGPU/sched-wmma-ds-load-window.mir b/llvm/test/CodeGen/AMDGPU/sched-wmma-ds-load-window.mir
new file mode 100644
index 0000000000000..6435dc89889e3
--- /dev/null
+++ b/llvm/test/CodeGen/AMDGPU/sched-wmma-ds-load-window.mir
@@ -0,0 +1,376 @@
+# RUN: llc -mtriple=amdgcn -mcpu=gfx1250 -run-pass=machine-scheduler -verify-misched -o - %s | FileCheck %s
+#
+# Test for the WMMASchedule DAG mutation on an 8x8 outer product tile
+# (64 DS_READ_B128, 8 A/B fragments, 64 V_WMMA). Without the mutation the
+# scheduler schedules all 64 loads at the beginning of the block; the mutation
+# debunches these loads.
+#
+# XFAIL: *
+
+--- |
+ target datalayout = "e-m:e-p:64:64-p1:64:64-p2:32:32-p3:32:32-p4:64:64-p5:32:32-p6:32:32-p7:160:256:256:32-p8:128:128:128:48-p9:192:256:256:32-i64:64-v16:16-v24:32-v32:32-v48:64-v96:128-v192:256-v256:256-v512:512-v1024:1024-v2048:2048-n32:64-S32-A5-G1-ni:7:8:9"
+ target triple = "amdgcn"
+ define amdgpu_kernel void @wmma_op() #0 { ret void }
+ attributes #0 = { "target-cpu"="gfx1250" "amdgpu-flat-work-group-size"="1,128" "amdgpu-waves-per-eu"="1,1" }
+...
+---
+name: wmma_op
+alignment: 4
+tracksRegLiveness: true
+isSSA: false
+noPhis: true
+machineFunctionInfo:
+ isEntryFunction: true
+ scratchRSrcReg: '$sgpr0_sgpr1_sgpr2_sgpr3'
+ stackPtrOffsetReg: '$sgpr32'
+ occupancy: 1
+body: |
+ bb.0:
+ ; CHECK-LABEL: name: wmma_op
+ ; CHECK: V_WMMA
+ ; CHECK-NEXT: V_WMMA
+ ; CHECK-NEXT: DS_READ_B128
+ ; CHECK-NEXT: DS_READ_B128
+ ; CHECK-NEXT: DS_READ_B128
+ ; CHECK-NEXT: DS_READ_B128
+ ; CHECK-NEXT: V_WMMA
+ ; CHECK-NEXT: DS_READ_B128
+ ; CHECK-NEXT: DS_READ_B128
+ ; CHECK-NEXT: DS_READ_B128
+ ; CHECK-NEXT: DS_READ_B128
+ ; CHECK-NEXT: V_WMMA
+ successors: %bb.1(0x80000000)
+ %ptra:vgpr_32 = IMPLICIT_DEF
+ %ptrb:vgpr_32 = IMPLICIT_DEF
+ %scale:vreg_64_lo256_align2 = IMPLICIT_DEF
+ %vo:vgpr_32 = IMPLICIT_DEF
+ %rsrc:sgpr_128 = IMPLICIT_DEF
+ %500:vreg_256_align2 = IMPLICIT_DEF
+ %501:vreg_256_align2 = IMPLICIT_DEF
+ %502:vreg_256_align2 = IMPLICIT_DEF
+ %503:vreg_256_align2 = IMPLICIT_DEF
+ %504:vreg_256_align2 = IMPLICIT_DEF
+ %505:vreg_256_align2 = IMPLICIT_DEF
+ %506:vreg_256_align2 = IMPLICIT_DEF
+ %507:vreg_256_align2 = IMPLICIT_DEF
+ %508:vreg_256_align2 = IMPLICIT_DEF
+ %509:vreg_256_align2 = IMPLICIT_DEF
+ %510:vreg_256_align2 = IMPLICIT_DEF
+ %511:vreg_256_align2 = IMPLICIT_DEF
+ %512:vreg_256_align2 = IMPLICIT_DEF
+ %513:vreg_256_align2 = IMPLICIT_DEF
+ %514:vreg_256_align2 = IMPLICIT_DEF
+ %515:vreg_256_align2 = IMPLICIT_DEF
+ %516:vreg_256_align2 = IMPLICIT_DEF
+ %517:vreg_256_align2 = IMPLICIT_DEF
+ %518:vreg_256_align2 = IMPLICIT_DEF
+ %519:vreg_256_align2 = IMPLICIT_DEF
+ %520:vreg_256_align2 = IMPLICIT_DEF
+ %521:vreg_256_align2 = IMPLICIT_DEF
+ %522:vreg_256_align2 = IMPLICIT_DEF
+ %523:vreg_256_align2 = IMPLICIT_DEF
+ %524:vreg_256_align2 = IMPLICIT_DEF
+ %525:vreg_256_align2 = IMPLICIT_DEF
+ %526:vreg_256_align2 = IMPLICIT_DEF
+ %527:vreg_256_align2 = IMPLICIT_DEF
+ %528:vreg_256_align2 = IMPLICIT_DEF
+ %529:vreg_256_align2 = IMPLICIT_DEF
+ %530:vreg_256_align2 = IMPLICIT_DEF
+ %531:vreg_256_align2 = IMPLICIT_DEF
+ %532:vreg_256_align2 = IMPLICIT_DEF
+ %533:vreg_256_align2 = IMPLICIT_DEF
+ %534:vreg_256_align2 = IMPLICIT_DEF
+ %535:vreg_256_align2 = IMPLICIT_DEF
+ %536:vreg_256_align2 = IMPLICIT_DEF
+ %537:vreg_256_align2 = IMPLICIT_DEF
+ %538:vreg_256_align2 = IMPLICIT_DEF
+ %539:vreg_256_align2 = IMPLICIT_DEF
+ %540:vreg_256_align2 = IMPLICIT_DEF
+ %541:vreg_256_align2 = IMPLICIT_DEF
+ %542:vreg_256_align2 = IMPLICIT_DEF
+ %543:vreg_256_align2 = IMPLICIT_DEF
+ %544:vreg_256_align2 = IMPLICIT_DEF
+ %545:vreg_256_align2 = IMPLICIT_DEF
+ %546:vreg_256_align2 = IMPLICIT_DEF
+ %547:vreg_256_align2 = IMPLICIT_DEF
+ %548:vreg_256_align2 = IMPLICIT_DEF
+ %549:vreg_256_align2 = IMPLICIT_DEF
+ %550:vreg_256_align2 = IMPLICIT_DEF
+ %551:vreg_256_align2 = IMPLICIT_DEF
+ %552:vreg_256_align2 = IMPLICIT_DEF
+ %553:vreg_256_align2 = IMPLICIT_DEF
+ %554:vreg_256_align2 = IMPLICIT_DEF
+ %555:vreg_256_align2 = IMPLICIT_DEF
+ %556:vreg_256_align2 = IMPLICIT_DEF
+ %557:vreg_256_align2 = IMPLICIT_DEF
+ %558:vreg_256_align2 = IMPLICIT_DEF
+ %559:vreg_256_align2 = IMPLICIT_DEF
+ %560:vreg_256_align2 = IMPLICIT_DEF
+ %561:vreg_256_align2 = IMPLICIT_DEF
+ %562:vreg_256_align2 = IMPLICIT_DEF
+ %563:vreg_256_align2 = IMPLICIT_DEF
+ S_BRANCH %bb.1
+
+ bb.1:
+ successors: %bb.2(0x80000000)
+ undef %300.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %300.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %300.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %300.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %301.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %301.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %301.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %301.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %302.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %302.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %302.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %302.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %303.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %303.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %303.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %303.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %304.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %304.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %304.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %304.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %305.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %305.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %305.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %305.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %306.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %306.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %306.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %306.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %307.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %307.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %307.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %307.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptra, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %400.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %400.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %400.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %400.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %401.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %401.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %401.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %401.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %402.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %402.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %402.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %402.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %403.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %403.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %403.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %403.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %404.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %404.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %404.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %404.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %405.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %405.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %405.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %405.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %406.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %406.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %406.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %406.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ undef %407.sub0_sub1_sub2_sub3:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 0, 0, implicit $exec :: (load (s128), addrspace 3)
+ %407.sub4_sub5_sub6_sub7:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 32, 0, implicit $exec :: (load (s128), addrspace 3)
+ %407.sub8_sub9_sub10_sub11:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 64, 0, implicit $exec :: (load (s128), addrspace 3)
+ %407.sub12_sub13_sub14_sub15:vreg_512_align2 = DS_READ_B128_gfx9 %ptrb, 96, 0, implicit $exec :: (load (s128), addrspace 3)
+ early-clobber %500:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %300, %400, 8, %500, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %501:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %300, %401, 8, %501, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %502:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %300, %402, 8, %502, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %503:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %300, %403, 8, %503, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %504:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %300, %404, 8, %504, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %505:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %300, %405, 8, %505, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %506:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %300, %406, 8, %506, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %507:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %300, %407, 8, %507, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %508:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %301, %400, 8, %508, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %509:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %301, %401, 8, %509, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %510:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %301, %402, 8, %510, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %511:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %301, %403, 8, %511, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %512:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %301, %404, 8, %512, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %513:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %301, %405, 8, %513, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %514:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %301, %406, 8, %514, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %515:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %301, %407, 8, %515, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %516:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %302, %400, 8, %516, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %517:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %302, %401, 8, %517, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %518:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %302, %402, 8, %518, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %519:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %302, %403, 8, %519, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %520:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %302, %404, 8, %520, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %521:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %302, %405, 8, %521, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %522:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %302, %406, 8, %522, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %523:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %302, %407, 8, %523, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %524:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %303, %400, 8, %524, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %525:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %303, %401, 8, %525, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %526:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %303, %402, 8, %526, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %527:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %303, %403, 8, %527, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %528:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %303, %404, 8, %528, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %529:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %303, %405, 8, %529, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %530:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %303, %406, 8, %530, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %531:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %303, %407, 8, %531, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %532:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %304, %400, 8, %532, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %533:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %304, %401, 8, %533, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %534:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %304, %402, 8, %534, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %535:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %304, %403, 8, %535, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %536:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %304, %404, 8, %536, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %537:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %304, %405, 8, %537, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %538:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %304, %406, 8, %538, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %539:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %304, %407, 8, %539, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %540:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %305, %400, 8, %540, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %541:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %305, %401, 8, %541, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %542:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %305, %402, 8, %542, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %543:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %305, %403, 8, %543, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %544:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %305, %404, 8, %544, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %545:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %305, %405, 8, %545, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %546:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %305, %406, 8, %546, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %547:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %305, %407, 8, %547, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %548:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %306, %400, 8, %548, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %549:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %306, %401, 8, %549, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %550:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %306, %402, 8, %550, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %551:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %306, %403, 8, %551, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %552:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %306, %404, 8, %552, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %553:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %306, %405, 8, %553, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %554:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %306, %406, 8, %554, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %555:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %306, %407, 8, %555, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %556:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %307, %400, 8, %556, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %557:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %307, %401, 8, %557, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %558:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %307, %402, 8, %558, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %559:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %307, %403, 8, %559, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %560:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %307, %404, 8, %560, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %561:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %307, %405, 8, %561, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %562:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %307, %406, 8, %562, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ early-clobber %563:vreg_256_align2 = V_WMMA_SCALE_F32_16X16X128_F8F6F4_f8_f8_w32_twoaddr %307, %407, 8, %563, %scale.sub0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, implicit $exec
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %500.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %500.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %501.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %501.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %502.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %502.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %503.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %503.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %504.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %504.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %505.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %505.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %506.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %506.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %507.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %507.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %508.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %508.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %509.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %509.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %510.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %510.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %511.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %511.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %512.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %512.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %513.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %513.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %514.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %514.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %515.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %515.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %516.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %516.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %517.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %517.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %518.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %518.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %519.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %519.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %520.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %520.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %521.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %521.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %522.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %522.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %523.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %523.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %524.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %524.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %525.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %525.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %526.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %526.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %527.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %527.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %528.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %528.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %529.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %529.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %530.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %530.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %531.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %531.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %532.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %532.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %533.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %533.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %534.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %534.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %535.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %535.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %536.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %536.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %537.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %537.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %538.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %538.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %539.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %539.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %540.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %540.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %541.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %541.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %542.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %542.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %543.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %543.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %544.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %544.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %545.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %545.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %546.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %546.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %547.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %547.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %548.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %548.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %549.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %549.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %550.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %550.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %551.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %551.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %552.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %552.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %553.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %553.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %554.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %554.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %555.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %555.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %556.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %556.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %557.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %557.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %558.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %558.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %559.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %559.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %560.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %560.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %561.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %561.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %562.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %562.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %563.sub0_sub1_sub2_sub3, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ BUFFER_STORE_DWORDX4_VBUFFER_OFFEN_exact %563.sub4_sub5_sub6_sub7, %vo, %rsrc, $sgpr_null, 0, 0, 0, implicit $exec :: (store (s128), addrspace 8)
+ S_BRANCH %bb.2
+
+ bb.2:
+ S_ENDPGM 0
+...
>From 5b3917deb6f2797469f70b7bc9f43dbdb6a5faaf Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Thu, 11 Jun 2026 21:08:50 +0000
Subject: [PATCH 05/20] Ordering WMMAs
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 10 ++++++++++
1 file changed, 10 insertions(+)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index fba0109a2fcca..897b1c20642f6 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -90,6 +90,16 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (Dist < 1)
Dist = 1;
+ // Ensure ordering of WMMAs.
+ auto [PrevSU, _] = *Wmmas.begin();
+ for (auto *It = std::next(Wmmas.begin()); It != Wmmas.end(); ++It) {
+ auto [SU, _] = *It;
+ bool Success = DAG->addEdge(SU, SDep(PrevSU, SDep::Artificial));
+ LLVM_DEBUG(dbgs() << "wmma SU" << SU->NodeNum << " <- after wmma SU"
+ << PrevSU->NodeNum << (Success ? "\n" : " FAIL (cycle)\n"));
+ PrevSU = SU;
+ }
+
// For each load, find the earliest and latest consuming WMMA positions.
for (LoadInfo& LI : Loads) {
for (const SDep &D : LI.SU->Succs) {
>From 53545a9a8488df979f7669251b2360d09ad3cbcc Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Thu, 11 Jun 2026 22:12:31 +0000
Subject: [PATCH 06/20] Bandwidth bounded ds_loads (keeping ds_loads from being
too close)
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 51 +++++++++++++++----
1 file changed, 41 insertions(+), 10 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 897b1c20642f6..809704a1ead73 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -34,6 +34,8 @@ struct LoadInfo {
SUnit *SU;
unsigned MinPos = UINT_MAX; // earliest consumer (UINT_MAX means none in this region)
unsigned MaxPos = 0; // latest consumer
+ long LatestCycle = 0; // Latest cycle this load can be scheduled for
+ bool BandwidthBound = false; // If bandwidth bound, it'll be placest at LatestCycle or earlier. Else, MaxPos.
};
class WMMASchedule : public ScheduleDAGMutation {
@@ -59,6 +61,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
MapVector<SUnit *, unsigned> Wmmas; // WMMA SUnit and its program position relative to one another
SmallVector<LoadInfo> Loads;
std::optional<unsigned> LoadLatency, WmmaLatency;
+ std::optional<double> LDSBandwidth;
for (SUnit &SU : DAG->SUnits) {
MachineInstr *MI = SU.getInstr();
@@ -75,21 +78,18 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
// Gather DS_LOADs
if (TII->isDS(*MI) && MI->mayLoad()) {
- if (!LoadLatency)
- LoadLatency = SM->computeInstrLatency(MI);
+ if (!LoadLatency) {
+ LoadLatency = SM->computeInstrLatency(MI); // TODO: Hardcode this.
+ LDSBandwidth = std::ceil(SM->computeReciprocalThroughput(MI)); // TODO: Possibly hardcode this?
+ }
Loads.push_back({&SU});
}
}
// The following means the DAG Mutation cannot do anything useful.
- if (!LoadLatency || !WmmaLatency || Wmmas.empty())
+ if (!LoadLatency || !LDSBandwidth || !WmmaLatency || Wmmas.empty())
return;
- // The number of WMMAs that elapse during one load's latency
- unsigned Dist = std::ceil(static_cast<double>(*LoadLatency) / *WmmaLatency);
- if (Dist < 1)
- Dist = 1;
-
// Ensure ordering of WMMAs.
auto [PrevSU, _] = *Wmmas.begin();
for (auto *It = std::next(Wmmas.begin()); It != Wmmas.end(); ++It) {
@@ -114,6 +114,11 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
}
}
+ // Order the Loads
+ llvm::stable_sort(Loads, [](const LoadInfo &A, const LoadInfo &B) {
+ return A.MinPos < B.MinPos;
+ });
+
// MaxPos in order
std::vector<bool> Present(Wmmas.size(), false);
for (const LoadInfo &LI : Loads)
@@ -131,17 +136,43 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
DeadBy[Pos] = Latest;
}
+ // For each load, determine if it needs to be bandwidth bound to prevent
+ // being too close to other loads.
+ for (LoadInfo& LI : Loads) {
+ if (LI.MinPos != UINT_MAX)
+ // Same thing as MinPos, but in cycles
+ LI.LatestCycle = (long)(LI.MinPos) * (*WmmaLatency) - (long)(*LoadLatency);
+ }
+ long PrevLatest = LONG_MAX;
+ for(int Pos = static_cast<int>(Loads.size()) - 1; Pos >= 0; --Pos) {
+ LoadInfo& LI = Loads[Pos];
+ if (LI.MinPos == UINT_MAX)
+ continue;
+ long Spaced = PrevLatest - (long)(*LDSBandwidth);
+ // Clamp to Spaced if the LatestCycle encroaches too close to another load
+ if (Spaced < LI.LatestCycle) {
+ LI.LatestCycle = Spaced;
+ LI.BandwidthBound = true;
+ }
+ PrevLatest = LI.LatestCycle;
+ }
+
// Create the edges that constrain where the ds_loads can be placed
// minimum distance of a load is LI.MinPos - Dist
// maximum distance of a load is DeadBy[LI.MinPos - Dist]
for (const LoadInfo &LI : Loads) {
- if (LI.MinPos == UINT_MAX || LI.MinPos < Dist)
+ if (LI.MinPos == UINT_MAX)
continue;
// Minimum distance edge
// Load scheduled before this point
- unsigned MinDist = LI.MinPos - Dist;
+ long MinDist = LI.LatestCycle / (long)(*WmmaLatency); // Go from cycle back to WMMA position (floor division)
+ if (MinDist < 0)
+ continue;
bool MinSuccess = DAG->addEdge(Wmmas.begin()[MinDist].first, SDep(LI.SU, SDep::Artificial));
+ LLVM_DEBUG(dbgs() << (LI.BandwidthBound ? "[bw] " : "[lat] ")
+ << "load SU" << LI.SU->NodeNum << " before W[" << MinDist
+ << "] (MinPos=" << LI.MinPos << " LatestCycle=" << LI.LatestCycle << ")\n");
// Maximum distance edge
// Load scheduled after this point
>From 0ddd465538eb229ed46b7dd28524a4835bc5057e Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Fri, 12 Jun 2026 00:00:50 +0000
Subject: [PATCH 07/20] DAG mutation now relies on latencies placed on edges.
Added ds_load->ds_load edges with hardcoded latency. Modified
ds_load->earliest wmma consumer edges to use hardcoded latency instead.
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 76 +++++++++----------
1 file changed, 37 insertions(+), 39 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 809704a1ead73..8729c117bf920 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -34,8 +34,6 @@ struct LoadInfo {
SUnit *SU;
unsigned MinPos = UINT_MAX; // earliest consumer (UINT_MAX means none in this region)
unsigned MaxPos = 0; // latest consumer
- long LatestCycle = 0; // Latest cycle this load can be scheduled for
- bool BandwidthBound = false; // If bandwidth bound, it'll be placest at LatestCycle or earlier. Else, MaxPos.
};
class WMMASchedule : public ScheduleDAGMutation {
@@ -60,7 +58,8 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
// Gather WMMAs (numbered in program order) and ds_loads.
MapVector<SUnit *, unsigned> Wmmas; // WMMA SUnit and its program position relative to one another
SmallVector<LoadInfo> Loads;
- std::optional<unsigned> LoadLatency, WmmaLatency;
+ std::optional<unsigned> LoadLatency = 64;
+ std::optional<unsigned> WmmaLatency;
std::optional<double> LDSBandwidth;
for (SUnit &SU : DAG->SUnits) {
@@ -78,10 +77,10 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
// Gather DS_LOADs
if (TII->isDS(*MI) && MI->mayLoad()) {
- if (!LoadLatency) {
- LoadLatency = SM->computeInstrLatency(MI); // TODO: Hardcode this.
- LDSBandwidth = std::ceil(SM->computeReciprocalThroughput(MI)); // TODO: Possibly hardcode this?
- }
+ if (!LoadLatency)
+ LoadLatency = SM->computeInstrLatency(MI);
+ if (!LDSBandwidth)
+ LDSBandwidth = std::ceil(SM->computeReciprocalThroughput(MI));
Loads.push_back({&SU});
}
}
@@ -90,7 +89,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (!LoadLatency || !LDSBandwidth || !WmmaLatency || Wmmas.empty())
return;
- // Ensure ordering of WMMAs.
+ // Order the WMMAs.
auto [PrevSU, _] = *Wmmas.begin();
for (auto *It = std::next(Wmmas.begin()); It != Wmmas.end(); ++It) {
auto [SU, _] = *It;
@@ -101,6 +100,8 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
}
// For each load, find the earliest and latest consuming WMMA positions.
+ // Additionally correct the latency of the ds_load -> earliest WMMA consumer
+ // data edge.
for (LoadInfo& LI : Loads) {
for (const SDep &D : LI.SU->Succs) {
if (D.getKind() != SDep::Data)
@@ -112,12 +113,35 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
LI.MinPos = std::min(LI.MinPos, It->second);
LI.MaxPos = std::max(LI.MaxPos, It->second);
}
+ if (LI.MinPos == UINT_MAX)
+ continue;
+ SUnit *EarliestConsumer = Wmmas.begin()[LI.MinPos].first;
+ // Correct latency of edges between ds_load and earliest WMMA consumer
+ for (SDep &S : LI.SU->Succs)
+ if (S.getSUnit() == EarliestConsumer && S.getKind() == SDep::Data)
+ S.setLatency(*LoadLatency);
+ for (SDep &P : EarliestConsumer->Preds)
+ if (P.getSUnit() == LI.SU && P.getKind() == SDep::Data)
+ P.setLatency(*LoadLatency);
+ EarliestConsumer->setDepthDirty();
+ LI.SU->setHeightDirty();
}
// Order the Loads
llvm::stable_sort(Loads, [](const LoadInfo &A, const LoadInfo &B) {
return A.MinPos < B.MinPos;
});
+ SUnit* Prev = nullptr;
+ for (LoadInfo &LI : Loads) {
+ if (LI.MinPos == UINT_MAX)
+ continue;
+ if (Prev) {
+ SDep D(Prev, SDep::Artificial);
+ D.setLatency(*LDSBandwidth);
+ DAG->addEdge(LI.SU, D);
+ }
+ Prev = LI.SU;
+ }
// MaxPos in order
std::vector<bool> Present(Wmmas.size(), false);
@@ -136,43 +160,17 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
DeadBy[Pos] = Latest;
}
- // For each load, determine if it needs to be bandwidth bound to prevent
- // being too close to other loads.
- for (LoadInfo& LI : Loads) {
- if (LI.MinPos != UINT_MAX)
- // Same thing as MinPos, but in cycles
- LI.LatestCycle = (long)(LI.MinPos) * (*WmmaLatency) - (long)(*LoadLatency);
- }
- long PrevLatest = LONG_MAX;
- for(int Pos = static_cast<int>(Loads.size()) - 1; Pos >= 0; --Pos) {
- LoadInfo& LI = Loads[Pos];
- if (LI.MinPos == UINT_MAX)
- continue;
- long Spaced = PrevLatest - (long)(*LDSBandwidth);
- // Clamp to Spaced if the LatestCycle encroaches too close to another load
- if (Spaced < LI.LatestCycle) {
- LI.LatestCycle = Spaced;
- LI.BandwidthBound = true;
- }
- PrevLatest = LI.LatestCycle;
- }
-
// Create the edges that constrain where the ds_loads can be placed
// minimum distance of a load is LI.MinPos - Dist
// maximum distance of a load is DeadBy[LI.MinPos - Dist]
+ unsigned Dist = std::max(1u, static_cast<unsigned>(std::ceil(static_cast<double>(*LoadLatency) / *WmmaLatency)));
for (const LoadInfo &LI : Loads) {
- if (LI.MinPos == UINT_MAX)
+ if (LI.MinPos == UINT_MAX || LI.MinPos < Dist)
continue;
// Minimum distance edge
- // Load scheduled before this point
- long MinDist = LI.LatestCycle / (long)(*WmmaLatency); // Go from cycle back to WMMA position (floor division)
- if (MinDist < 0)
- continue;
- bool MinSuccess = DAG->addEdge(Wmmas.begin()[MinDist].first, SDep(LI.SU, SDep::Artificial));
- LLVM_DEBUG(dbgs() << (LI.BandwidthBound ? "[bw] " : "[lat] ")
- << "load SU" << LI.SU->NodeNum << " before W[" << MinDist
- << "] (MinPos=" << LI.MinPos << " LatestCycle=" << LI.LatestCycle << ")\n");
+ // Load scheduled before this point (heuristically)
+ long MinDist = LI.MinPos - Dist;
// Maximum distance edge
// Load scheduled after this point
@@ -183,7 +181,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
LLVM_DEBUG(dbgs() << "load SU" << LI.SU->NodeNum << " Win=[" << (MaxDist ? *MaxDist : -1) << ","
<< MinDist << "] MinPos=" << LI.MinPos << " MaxPos=" << LI.MaxPos
- << (MinSuccess && MaxSuccess ? "\n" : " (edge REJECTED: cycle)\n"));
+ << (MaxSuccess ? "\n" : " (edge REJECTED: cycle)\n"));
}
}
>From 3cdc891ea898ace867cbbdb95e0a5025cdb63e35 Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Mon, 15 Jun 2026 02:42:33 +0000
Subject: [PATCH 08/20] Determine earliest point a ds_load should be loaded by
calculating overlapping live ranges
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 170 ++++++++++++------
1 file changed, 113 insertions(+), 57 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 8729c117bf920..85ea68f4ebf62 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -6,13 +6,24 @@
//
//===----------------------------------------------------------------------===//
//
-/// \file This file contains a DAG scheduling mutation to add additional
-/// edges between ds_load instructions and wmma instructions that
-/// occur a certain amount away from the actual wmma consumer of
-/// said ds_load. This forces the ds_load to properly prefetch
-/// and prevent early bunching of ds_loads that then lead to long
-/// stalls.
-//
+/// \file This file contains a DAG scheduling mutation that shapes how gfx1250
+/// ds_load (LDS) prefetches are placed relative to the WMMA instructions
+/// that consume them, to prevent the pre-RA scheduler from bunching all
+/// the loads at the head of the block (which forces the WMMAs behind
+/// long s_wait_dscnt stalls and inflates register pressure).
+///
+/// It does the following:
+/// - Order the WMMAs (WMMA -> WMMA edges added).
+/// - Order the ds_loads (ds_load -> ds_load edges added
+/// with latency attached to prevent them overhwelming
+/// the LDS bus and becoming memory bound).
+/// - Add WMMA -> ds_load edges to stop loads from being bunched at
+/// the start of the block
+/// - Build a live range histogram of the A/B operand fragments under
+/// an as late as possible schedule, recording the minimum VGPRs
+/// needed for such a schedule (so the WMMA -> ds_load edges can
+/// be placed earlier if the minimum VGPR budget can afford it).
+///
//===----------------------------------------------------------------------===//
#include "AMDGPUWMMASchedule.h"
@@ -20,7 +31,6 @@
#include "SIInstrInfo.h"
#include "llvm/CodeGen/ScheduleDAG.h"
#include "llvm/CodeGen/ScheduleDAGInstrs.h"
-#include "llvm/Support/Debug.h"
#include <optional>
#define DEBUG_TYPE "amdgpu-wmma-sched"
@@ -28,12 +38,24 @@ using namespace llvm;
namespace {
-// A ds_load plus the program order positions (among WMMAs) of its
-// earliest and latest WMMA consumer.
+// A single ds_load and their order among the WMMAs.
struct LoadInfo {
SUnit *SU;
- unsigned MinPos = UINT_MAX; // earliest consumer (UINT_MAX means none in this region)
- unsigned MaxPos = 0; // latest consumer
+ unsigned MinPos = UINT_MAX; // earliest WMMA consumer (UINT_MAX means none in region)
+ unsigned MaxPos = 0; // latest WMMA consumer
+ long LatestCycle = 0; // as late as possible cycle
+};
+
+// A fragment: the wide vreg several ds_loads build (for example - a vreg_512
+// from four DS_READ_B128). This is the unit for VGPR pressure - the DS_READ
+// subloads share one register, so counting per ds_load instead of by fragments
+// would multiply the pressure.
+struct FragInfo {
+ unsigned VGPRs = 0;
+ unsigned MinPos = UINT_MAX;
+ unsigned MaxPos = 0;
+ long LatestCycle = LONG_MAX; // earliest subload's as late as possible cycle
+ SmallVector<SUnit *, 4> Subloads;
};
class WMMASchedule : public ScheduleDAGMutation {
@@ -56,9 +78,9 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
const SIInstrInfo *TII = ST.getInstrInfo();
// Gather WMMAs (numbered in program order) and ds_loads.
- MapVector<SUnit *, unsigned> Wmmas; // WMMA SUnit and its program position relative to one another
+ MapVector<SUnit *, unsigned> Wmmas; // Ordered WMMA SUnits
SmallVector<LoadInfo> Loads;
- std::optional<unsigned> LoadLatency = 64;
+ std::optional<unsigned> LoadLatency;
std::optional<unsigned> WmmaLatency;
std::optional<double> LDSBandwidth;
@@ -93,16 +115,14 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
auto [PrevSU, _] = *Wmmas.begin();
for (auto *It = std::next(Wmmas.begin()); It != Wmmas.end(); ++It) {
auto [SU, _] = *It;
- bool Success = DAG->addEdge(SU, SDep(PrevSU, SDep::Artificial));
- LLVM_DEBUG(dbgs() << "wmma SU" << SU->NodeNum << " <- after wmma SU"
- << PrevSU->NodeNum << (Success ? "\n" : " FAIL (cycle)\n"));
+ DAG->addEdge(SU, SDep(PrevSU, SDep::Artificial));
PrevSU = SU;
}
- // For each load, find the earliest and latest consuming WMMA positions.
- // Additionally correct the latency of the ds_load -> earliest WMMA consumer
- // data edge.
- for (LoadInfo& LI : Loads) {
+ // For each load, find earliest and latest consuming WMMA positions, and
+ // correct the ds_load -> earliest consumer data edge latency (Both the
+ // Succs and Preds SDep is updated)
+ for (LoadInfo &LI : Loads) {
for (const SDep &D : LI.SU->Succs) {
if (D.getKind() != SDep::Data)
continue;
@@ -131,7 +151,9 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
llvm::stable_sort(Loads, [](const LoadInfo &A, const LoadInfo &B) {
return A.MinPos < B.MinPos;
});
- SUnit* Prev = nullptr;
+
+ // Chain consecutive loads with an LDS bandwidth latency
+ SUnit *Prev = nullptr;
for (LoadInfo &LI : Loads) {
if (LI.MinPos == UINT_MAX)
continue;
@@ -143,47 +165,81 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
Prev = LI.SU;
}
- // MaxPos in order
- std::vector<bool> Present(Wmmas.size(), false);
- for (const LoadInfo &LI : Loads)
- // Check if there's a consumer for this load in the region
+ // Determing each load's as late as possible cycle - this means the
+ // latest cycle that still meets the load latency, then pushed earlier
+ // if the ds_load -> ds_load edges requires spacing (ds_loads cannot be
+ // too close to each other or it could overwhelm the LDS bus and lead to
+ // the program being memory bound).
+ for (LoadInfo &LI : Loads)
if (LI.MinPos != UINT_MAX)
- Present[LI.MaxPos] = true;
-
- // Latest position that's <= position Pos at which some load's register frees
- // A load's register becomes free at its MaxPos, which is its last WMMA consumer.
- std::vector<std::optional<unsigned>> DeadBy(Wmmas.size(), std::nullopt);
- std::optional<unsigned> Latest;
- for (unsigned Pos = 0; Pos < Wmmas.size(); ++Pos) {
- if (Present[Pos])
- Latest = Pos;
- DeadBy[Pos] = Latest;
+ LI.LatestCycle = (long)LI.MinPos * (*WmmaLatency) - (long)(*LoadLatency);
+ long PrevLatest = LONG_MAX;
+ for (int I = (int)Loads.size() - 1; I >= 0; --I) {
+ LoadInfo &LI = Loads[I];
+ if (LI.MinPos == UINT_MAX)
+ continue;
+ long Spaced = PrevLatest - (long)(*LDSBandwidth);
+ if (Spaced < LI.LatestCycle)
+ LI.LatestCycle = Spaced;
+ PrevLatest = LI.LatestCycle;
}
- // Create the edges that constrain where the ds_loads can be placed
- // minimum distance of a load is LI.MinPos - Dist
- // maximum distance of a load is DeadBy[LI.MinPos - Dist]
- unsigned Dist = std::max(1u, static_cast<unsigned>(std::ceil(static_cast<double>(*LoadLatency) / *WmmaLatency)));
- for (const LoadInfo &LI : Loads) {
- if (LI.MinPos == UINT_MAX || LI.MinPos < Dist)
+ // Group subloads into fragments and build the live range histogram
+ // with a schedule as late as possible. Each fragment is live from
+ // its earliest subload to its last WMMA consumer. The peak of the
+ // histogram is the minimum VGPRs needed.
+ MapVector<Register, FragInfo> Frags;
+ for (LoadInfo &LI : Loads) {
+ if (LI.MinPos == UINT_MAX)
continue;
-
- // Minimum distance edge
- // Load scheduled before this point (heuristically)
- long MinDist = LI.MinPos - Dist;
-
- // Maximum distance edge
- // Load scheduled after this point
- bool MaxSuccess = true;
- std::optional<unsigned> MaxDist = DeadBy[MinDist];
- if (MaxDist && *MaxDist < MinDist)
- MaxSuccess = DAG->addEdge(LI.SU, SDep(Wmmas.begin()[*MaxDist].first, SDep::Artificial));
-
- LLVM_DEBUG(dbgs() << "load SU" << LI.SU->NodeNum << " Win=[" << (MaxDist ? *MaxDist : -1) << ","
- << MinDist << "] MinPos=" << LI.MinPos << " MaxPos=" << LI.MaxPos
- << (MaxSuccess ? "\n" : " (edge REJECTED: cycle)\n"));
+ Register R = LI.SU->getInstr()->getOperand(0).getReg();
+ FragInfo &F = Frags[R];
+ if (F.Subloads.empty() && R.isVirtual())
+ F.VGPRs = TRI.getRegSizeInBits(*MRI.getRegClass(R)) / 32;
+ F.MinPos = std::min(F.MinPos, LI.MinPos);
+ F.MaxPos = std::max(F.MaxPos, LI.MaxPos);
+ F.LatestCycle = std::min(F.LatestCycle, LI.LatestCycle);
+ F.Subloads.push_back(LI.SU);
+ }
+
+ std::vector<unsigned> Hist(Wmmas.size(), 0);
+ for (auto &KV : Frags) {
+ FragInfo &F = KV.second;
+ long Pos = F.LatestCycle / (long)(*WmmaLatency);
+ unsigned StartPos = Pos < 0 ? 0 : (unsigned)Pos;
+ for (unsigned P = StartPos; P <= F.MaxPos && P < Wmmas.size(); ++P)
+ Hist[P] += F.VGPRs;
}
+ unsigned Budget = 0;
+ for (unsigned P = 0; P < Wmmas.size(); ++P)
+ Budget = std::max(Budget, Hist[P]);
+
+ // For each fragment (in order), find the earliest position it
+ // can be placed so the live set never exceeds the budget, then
+ // add a WMMAS[Earliest] -> ds_load edge - this is what leads to
+ // the debunching.
+ for (auto &KV : Frags) {
+ FragInfo &F = KV.second;
+ long Pos = F.LatestCycle / (long)(*WmmaLatency);
+ unsigned LateStartPos = Pos < 0 ? 0 : (unsigned)Pos;
+ unsigned Earliest = LateStartPos;
+ for (int P = (int)LateStartPos - 1; P >= 0; --P) {
+ if (Hist[(unsigned)P] + F.VGPRs <= Budget)
+ Earliest = (unsigned)P;
+ else
+ break;
+ }
+ // Update the histogram so later fragments don't schedule earlier and
+ // exceed the budget.
+ for (unsigned P = Earliest; P < LateStartPos; ++P)
+ Hist[P] += F.VGPRs;
+ // No need to add an edge if the load can be scheduled at the beginning.
+ if (Earliest == 0)
+ continue;
+ for (SUnit *L : F.Subloads)
+ DAG->addEdge(L, SDep(Wmmas.begin()[Earliest].first, SDep::Artificial));
+ }
}
} // end namespace
>From feedd8a5deb641be5c1e6564e8a789319372e073 Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Mon, 15 Jun 2026 07:42:01 +0000
Subject: [PATCH 09/20] clang format
---
.../lib/Target/AMDGPU/AMDGPUTargetMachine.cpp | 209 +++++++++---------
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 23 +-
2 files changed, 112 insertions(+), 120 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
index 48673fc0d2ad1..6236959bd136a 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
@@ -185,13 +185,13 @@ class AMDGPUCodeGenPassBuilder
class SGPRRegisterRegAlloc : public RegisterRegAllocBase<SGPRRegisterRegAlloc> {
public:
SGPRRegisterRegAlloc(const char *N, const char *D, FunctionPassCtor C)
- : RegisterRegAllocBase(N, D, C) {}
+ : RegisterRegAllocBase(N, D, C) {}
};
class VGPRRegisterRegAlloc : public RegisterRegAllocBase<VGPRRegisterRegAlloc> {
public:
VGPRRegisterRegAlloc(const char *N, const char *D, FunctionPassCtor C)
- : RegisterRegAllocBase(N, D, C) {}
+ : RegisterRegAllocBase(N, D, C) {}
};
class WWMRegisterRegAlloc : public RegisterRegAllocBase<WWMRegisterRegAlloc> {
@@ -234,19 +234,21 @@ static llvm::once_flag InitializeDefaultVGPRRegisterAllocatorFlag;
static llvm::once_flag InitializeDefaultWWMRegisterAllocatorFlag;
static SGPRRegisterRegAlloc
-defaultSGPRRegAlloc("default",
- "pick SGPR register allocator based on -O option",
- useDefaultRegisterAllocator);
+ defaultSGPRRegAlloc("default",
+ "pick SGPR register allocator based on -O option",
+ useDefaultRegisterAllocator);
static cl::opt<SGPRRegisterRegAlloc::FunctionPassCtor, false,
RegisterPassParser<SGPRRegisterRegAlloc>>
-SGPRRegAlloc("sgpr-regalloc", cl::Hidden, cl::init(&useDefaultRegisterAllocator),
- cl::desc("Register allocator to use for SGPRs"));
+ SGPRRegAlloc("sgpr-regalloc", cl::Hidden,
+ cl::init(&useDefaultRegisterAllocator),
+ cl::desc("Register allocator to use for SGPRs"));
static cl::opt<VGPRRegisterRegAlloc::FunctionPassCtor, false,
RegisterPassParser<VGPRRegisterRegAlloc>>
-VGPRRegAlloc("vgpr-regalloc", cl::Hidden, cl::init(&useDefaultRegisterAllocator),
- cl::desc("Register allocator to use for VGPRs"));
+ VGPRRegAlloc("vgpr-regalloc", cl::Hidden,
+ cl::init(&useDefaultRegisterAllocator),
+ cl::desc("Register allocator to use for VGPRs"));
static cl::opt<WWMRegisterRegAlloc::FunctionPassCtor, false,
RegisterPassParser<WWMRegisterRegAlloc>>
@@ -374,22 +376,25 @@ static FunctionPass *createFastWWMRegisterAllocator() {
return createFastRegisterAllocator(onlyAllocateWWMRegs, false);
}
-static SGPRRegisterRegAlloc basicRegAllocSGPR(
- "basic", "basic register allocator", createBasicSGPRRegisterAllocator);
-static SGPRRegisterRegAlloc greedyRegAllocSGPR(
- "greedy", "greedy register allocator", createGreedySGPRRegisterAllocator);
-
-static SGPRRegisterRegAlloc fastRegAllocSGPR(
- "fast", "fast register allocator", createFastSGPRRegisterAllocator);
+static SGPRRegisterRegAlloc basicRegAllocSGPR("basic",
+ "basic register allocator",
+ createBasicSGPRRegisterAllocator);
+static SGPRRegisterRegAlloc
+ greedyRegAllocSGPR("greedy", "greedy register allocator",
+ createGreedySGPRRegisterAllocator);
+static SGPRRegisterRegAlloc fastRegAllocSGPR("fast", "fast register allocator",
+ createFastSGPRRegisterAllocator);
-static VGPRRegisterRegAlloc basicRegAllocVGPR(
- "basic", "basic register allocator", createBasicVGPRRegisterAllocator);
-static VGPRRegisterRegAlloc greedyRegAllocVGPR(
- "greedy", "greedy register allocator", createGreedyVGPRRegisterAllocator);
+static VGPRRegisterRegAlloc basicRegAllocVGPR("basic",
+ "basic register allocator",
+ createBasicVGPRRegisterAllocator);
+static VGPRRegisterRegAlloc
+ greedyRegAllocVGPR("greedy", "greedy register allocator",
+ createGreedyVGPRRegisterAllocator);
-static VGPRRegisterRegAlloc fastRegAllocVGPR(
- "fast", "fast register allocator", createFastVGPRRegisterAllocator);
+static VGPRRegisterRegAlloc fastRegAllocVGPR("fast", "fast register allocator",
+ createFastVGPRRegisterAllocator);
static WWMRegisterRegAlloc basicRegAllocWWMReg("basic",
"basic register allocator",
createBasicWWMRegisterAllocator);
@@ -406,14 +411,14 @@ static bool isLTOPreLink(ThinOrFullLTOPhase Phase) {
} // anonymous namespace
static cl::opt<bool>
-EnableEarlyIfConversion("amdgpu-early-ifcvt", cl::Hidden,
- cl::desc("Run early if-conversion"),
- cl::init(false));
+ EnableEarlyIfConversion("amdgpu-early-ifcvt", cl::Hidden,
+ cl::desc("Run early if-conversion"),
+ cl::init(false));
static cl::opt<bool>
-OptExecMaskPreRA("amdgpu-opt-exec-mask-pre-ra", cl::Hidden,
- cl::desc("Run pre-RA exec mask optimizations"),
- cl::init(true));
+ OptExecMaskPreRA("amdgpu-opt-exec-mask-pre-ra", cl::Hidden,
+ cl::desc("Run pre-RA exec mask optimizations"),
+ cl::init(true));
static cl::opt<bool>
LowerCtorDtor("amdgpu-lower-global-ctor-dtor",
@@ -421,32 +426,27 @@ static cl::opt<bool>
cl::init(true), cl::Hidden);
// Option to disable vectorizer for tests.
-static cl::opt<bool> EnableLoadStoreVectorizer(
- "amdgpu-load-store-vectorizer",
- cl::desc("Enable load store vectorizer"),
- cl::init(true),
- cl::Hidden);
+static cl::opt<bool>
+ EnableLoadStoreVectorizer("amdgpu-load-store-vectorizer",
+ cl::desc("Enable load store vectorizer"),
+ cl::init(true), cl::Hidden);
// Option to control global loads scalarization
-static cl::opt<bool> ScalarizeGlobal(
- "amdgpu-scalarize-global-loads",
- cl::desc("Enable global load scalarization"),
- cl::init(true),
- cl::Hidden);
+static cl::opt<bool>
+ ScalarizeGlobal("amdgpu-scalarize-global-loads",
+ cl::desc("Enable global load scalarization"),
+ cl::init(true), cl::Hidden);
// Option to run internalize pass.
static cl::opt<bool> InternalizeSymbols(
- "amdgpu-internalize-symbols",
- cl::desc("Enable elimination of non-kernel functions and unused globals"),
- cl::init(false),
- cl::Hidden);
+ "amdgpu-internalize-symbols",
+ cl::desc("Enable elimination of non-kernel functions and unused globals"),
+ cl::init(false), cl::Hidden);
// Option to inline all early.
-static cl::opt<bool> EarlyInlineAll(
- "amdgpu-early-inline-all",
- cl::desc("Inline all functions early"),
- cl::init(false),
- cl::Hidden);
+static cl::opt<bool> EarlyInlineAll("amdgpu-early-inline-all",
+ cl::desc("Inline all functions early"),
+ cl::init(false), cl::Hidden);
static cl::opt<bool> RemoveIncompatibleFunctions(
"amdgpu-enable-remove-incompatible-functions", cl::Hidden,
@@ -454,39 +454,35 @@ static cl::opt<bool> RemoveIncompatibleFunctions(
"use features not supported by the target GPU"),
cl::init(true));
-static cl::opt<bool> EnableSDWAPeephole(
- "amdgpu-sdwa-peephole",
- cl::desc("Enable SDWA peepholer"),
- cl::init(true));
+static cl::opt<bool> EnableSDWAPeephole("amdgpu-sdwa-peephole",
+ cl::desc("Enable SDWA peepholer"),
+ cl::init(true));
-static cl::opt<bool> EnableDPPCombine(
- "amdgpu-dpp-combine",
- cl::desc("Enable DPP combiner"),
- cl::init(true));
+static cl::opt<bool> EnableDPPCombine("amdgpu-dpp-combine",
+ cl::desc("Enable DPP combiner"),
+ cl::init(true));
// Enable address space based alias analysis
-static cl::opt<bool> EnableAMDGPUAliasAnalysis("enable-amdgpu-aa", cl::Hidden,
- cl::desc("Enable AMDGPU Alias Analysis"),
- cl::init(true));
+static cl::opt<bool>
+ EnableAMDGPUAliasAnalysis("enable-amdgpu-aa", cl::Hidden,
+ cl::desc("Enable AMDGPU Alias Analysis"),
+ cl::init(true));
// Enable lib calls simplifications
-static cl::opt<bool> EnableLibCallSimplify(
- "amdgpu-simplify-libcall",
- cl::desc("Enable amdgpu library simplifications"),
- cl::init(true),
- cl::Hidden);
+static cl::opt<bool>
+ EnableLibCallSimplify("amdgpu-simplify-libcall",
+ cl::desc("Enable amdgpu library simplifications"),
+ cl::init(true), cl::Hidden);
static cl::opt<bool> EnableLowerKernelArguments(
- "amdgpu-ir-lower-kernel-arguments",
- cl::desc("Lower kernel argument loads in IR pass"),
- cl::init(true),
- cl::Hidden);
+ "amdgpu-ir-lower-kernel-arguments",
+ cl::desc("Lower kernel argument loads in IR pass"), cl::init(true),
+ cl::Hidden);
static cl::opt<bool> EnableRegReassign(
- "amdgpu-reassign-regs",
- cl::desc("Enable register reassign optimizations on gfx10+"),
- cl::init(true),
- cl::Hidden);
+ "amdgpu-reassign-regs",
+ cl::desc("Enable register reassign optimizations on gfx10+"),
+ cl::init(true), cl::Hidden);
static cl::opt<bool> OptVGPRLiveRange(
"amdgpu-opt-vgpr-liverange",
@@ -504,11 +500,10 @@ static cl::opt<ScanOptions> AMDGPUAtomicOptimizerStrategy(
clEnumValN(ScanOptions::None, "None", "Disable atomic optimizer")));
// Enable Mode register optimization
-static cl::opt<bool> EnableSIModeRegisterPass(
- "amdgpu-mode-register",
- cl::desc("Enable mode register pass"),
- cl::init(true),
- cl::Hidden);
+static cl::opt<bool>
+ EnableSIModeRegisterPass("amdgpu-mode-register",
+ cl::desc("Enable mode register pass"),
+ cl::init(true), cl::Hidden);
// Enable GFX11+ s_delay_alu insertion
static cl::opt<bool>
@@ -524,19 +519,16 @@ static cl::opt<bool>
// Option is used in lit tests to prevent deadcoding of patterns inspected.
static cl::opt<bool>
-EnableDCEInRA("amdgpu-dce-in-ra",
- cl::init(true), cl::Hidden,
- cl::desc("Enable machine DCE inside regalloc"));
+ EnableDCEInRA("amdgpu-dce-in-ra", cl::init(true), cl::Hidden,
+ cl::desc("Enable machine DCE inside regalloc"));
static cl::opt<bool> EnableSetWavePriority("amdgpu-set-wave-priority",
cl::desc("Adjust wave priority"),
cl::init(false), cl::Hidden);
-static cl::opt<bool> EnableScalarIRPasses(
- "amdgpu-scalar-ir-passes",
- cl::desc("Enable scalar IR passes"),
- cl::init(true),
- cl::Hidden);
+static cl::opt<bool> EnableScalarIRPasses("amdgpu-scalar-ir-passes",
+ cl::desc("Enable scalar IR passes"),
+ cl::init(true), cl::Hidden);
static cl::opt<bool> EnableLowerExecSync(
"amdgpu-enable-lower-exec-sync",
@@ -560,10 +552,10 @@ static cl::opt<bool, true> EnableLowerModuleLDS(
cl::location(AMDGPUTargetMachine::EnableLowerModuleLDS), cl::init(true),
cl::Hidden);
-static cl::opt<bool> EnablePreRAOptimizations(
- "amdgpu-enable-pre-ra-optimizations",
- cl::desc("Enable Pre-RA optimizations pass"), cl::init(true),
- cl::Hidden);
+static cl::opt<bool>
+ EnablePreRAOptimizations("amdgpu-enable-pre-ra-optimizations",
+ cl::desc("Enable Pre-RA optimizations pass"),
+ cl::init(true), cl::Hidden);
static cl::opt<bool> EnablePromoteKernelArguments(
"amdgpu-enable-promote-kernel-arguments",
@@ -622,10 +614,10 @@ static cl::opt<bool> EnableRewritePartialRegUses(
cl::desc("Enable rewrite partial reg uses pass"), cl::init(true),
cl::Hidden);
-static cl::opt<bool> EnableHipStdPar(
- "amdgpu-enable-hipstdpar",
- cl::desc("Enable HIP Standard Parallelism Offload support"), cl::init(false),
- cl::Hidden);
+static cl::opt<bool>
+ EnableHipStdPar("amdgpu-enable-hipstdpar",
+ cl::desc("Enable HIP Standard Parallelism Offload support"),
+ cl::init(false), cl::Hidden);
static cl::opt<bool>
EnableAMDGPUAttributor("amdgpu-attributor-enable",
@@ -751,8 +743,8 @@ static ScheduleDAGInstrs *createSIMachineScheduler(MachineSchedContext *C) {
static ScheduleDAGInstrs *
createGCNMaxOccupancyMachineScheduler(MachineSchedContext *C) {
const GCNSubtarget &ST = C->MF->getSubtarget<GCNSubtarget>();
- ScheduleDAGMILive *DAG =
- new GCNScheduleDAGMILive(C, std::make_unique<GCNMaxOccupancySchedStrategy>(C));
+ ScheduleDAGMILive *DAG = new GCNScheduleDAGMILive(
+ C, std::make_unique<GCNMaxOccupancySchedStrategy>(C));
DAG->addMutation(createLoadClusterDAGMutation(DAG->TII, DAG->TRI));
if (ST.shouldClusterStores())
DAG->addMutation(createStoreClusterDAGMutation(DAG->TII, DAG->TRI));
@@ -818,14 +810,13 @@ createIterativeILPMachineScheduler(MachineSchedContext *C) {
return DAG;
}
-static MachineSchedRegistry
-SISchedRegistry("si", "Run SI's custom scheduler",
- createSIMachineScheduler);
+static MachineSchedRegistry SISchedRegistry("si", "Run SI's custom scheduler",
+ createSIMachineScheduler);
static MachineSchedRegistry
-GCNMaxOccupancySchedRegistry("gcn-max-occupancy",
- "Run GCN scheduler to maximize occupancy",
- createGCNMaxOccupancyMachineScheduler);
+ GCNMaxOccupancySchedRegistry("gcn-max-occupancy",
+ "Run GCN scheduler to maximize occupancy",
+ createGCNMaxOccupancyMachineScheduler);
static MachineSchedRegistry
GCNMaxILPSchedRegistry("gcn-max-ilp", "Run GCN scheduler to maximize ilp",
@@ -1520,7 +1511,7 @@ void AMDGPUPassConfig::addIRPasses() {
AAResults &AAR) {
if (auto *WrapperPass = P.getAnalysisIfAvailable<AMDGPUAAWrapperPass>())
AAR.addAAResult(WrapperPass->getResult());
- }));
+ }));
}
if (TM.getTargetTriple().isAMDGCN()) {
@@ -2134,8 +2125,8 @@ bool GCNTargetMachine::parseMachineFunctionInfo(
AMDGPU::SGPR_32RegClass,
MFI->ArgInfo.PrivateSegmentSize, 0, 0) ||
parseAndCheckArgument(YamlMFI.ArgInfo->LDSKernelId,
- AMDGPU::SGPR_32RegClass,
- MFI->ArgInfo.LDSKernelId, 0, 1) ||
+ AMDGPU::SGPR_32RegClass, MFI->ArgInfo.LDSKernelId,
+ 0, 1) ||
parseAndCheckArgument(YamlMFI.ArgInfo->WorkGroupIDX,
AMDGPU::SGPR_32RegClass, MFI->ArgInfo.WorkGroupIDX,
0, 1) ||
@@ -2158,14 +2149,14 @@ bool GCNTargetMachine::parseMachineFunctionInfo(
AMDGPU::SReg_64RegClass,
MFI->ArgInfo.ImplicitBufferPtr, 2, 0) ||
parseAndCheckArgument(YamlMFI.ArgInfo->WorkItemIDX,
- AMDGPU::VGPR_32RegClass,
- MFI->ArgInfo.WorkItemIDX, 0, 0) ||
+ AMDGPU::VGPR_32RegClass, MFI->ArgInfo.WorkItemIDX,
+ 0, 0) ||
parseAndCheckArgument(YamlMFI.ArgInfo->WorkItemIDY,
- AMDGPU::VGPR_32RegClass,
- MFI->ArgInfo.WorkItemIDY, 0, 0) ||
+ AMDGPU::VGPR_32RegClass, MFI->ArgInfo.WorkItemIDY,
+ 0, 0) ||
parseAndCheckArgument(YamlMFI.ArgInfo->WorkItemIDZ,
- AMDGPU::VGPR_32RegClass,
- MFI->ArgInfo.WorkItemIDZ, 0, 0)))
+ AMDGPU::VGPR_32RegClass, MFI->ArgInfo.WorkItemIDZ,
+ 0, 0)))
return true;
// Parse FirstKernArgPreloadReg separately, since it's a Register,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 85ea68f4ebf62..4cdd0f5d1ecca 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -20,7 +20,7 @@
/// - Add WMMA -> ds_load edges to stop loads from being bunched at
/// the start of the block
/// - Build a live range histogram of the A/B operand fragments under
-/// an as late as possible schedule, recording the minimum VGPRs
+/// an as late as possible schedule, recording the minimum VGPRs
/// needed for such a schedule (so the WMMA -> ds_load edges can
/// be placed earlier if the minimum VGPR budget can afford it).
///
@@ -41,9 +41,10 @@ namespace {
// A single ds_load and their order among the WMMAs.
struct LoadInfo {
SUnit *SU;
- unsigned MinPos = UINT_MAX; // earliest WMMA consumer (UINT_MAX means none in region)
- unsigned MaxPos = 0; // latest WMMA consumer
- long LatestCycle = 0; // as late as possible cycle
+ unsigned MinPos =
+ UINT_MAX; // earliest WMMA consumer (UINT_MAX means none in region)
+ unsigned MaxPos = 0; // latest WMMA consumer
+ long LatestCycle = 0; // as late as possible cycle
};
// A fragment: the wide vreg several ds_loads build (for example - a vreg_512
@@ -54,7 +55,7 @@ struct FragInfo {
unsigned VGPRs = 0;
unsigned MinPos = UINT_MAX;
unsigned MaxPos = 0;
- long LatestCycle = LONG_MAX; // earliest subload's as late as possible cycle
+ long LatestCycle = LONG_MAX; // earliest subload's as late as possible cycle
SmallVector<SUnit *, 4> Subloads;
};
@@ -99,7 +100,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
// Gather DS_LOADs
if (TII->isDS(*MI) && MI->mayLoad()) {
- if (!LoadLatency)
+ if (!LoadLatency)
LoadLatency = SM->computeInstrLatency(MI);
if (!LDSBandwidth)
LDSBandwidth = std::ceil(SM->computeReciprocalThroughput(MI));
@@ -120,7 +121,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
}
// For each load, find earliest and latest consuming WMMA positions, and
- // correct the ds_load -> earliest consumer data edge latency (Both the
+ // correct the ds_load -> earliest consumer data edge latency (Both the
// Succs and Preds SDep is updated)
for (LoadInfo &LI : Loads) {
for (const SDep &D : LI.SU->Succs) {
@@ -133,7 +134,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
LI.MinPos = std::min(LI.MinPos, It->second);
LI.MaxPos = std::max(LI.MaxPos, It->second);
}
- if (LI.MinPos == UINT_MAX)
+ if (LI.MinPos == UINT_MAX)
continue;
SUnit *EarliestConsumer = Wmmas.begin()[LI.MinPos].first;
// Correct latency of edges between ds_load and earliest WMMA consumer
@@ -155,7 +156,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
// Chain consecutive loads with an LDS bandwidth latency
SUnit *Prev = nullptr;
for (LoadInfo &LI : Loads) {
- if (LI.MinPos == UINT_MAX)
+ if (LI.MinPos == UINT_MAX)
continue;
if (Prev) {
SDep D(Prev, SDep::Artificial);
@@ -185,8 +186,8 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
}
// Group subloads into fragments and build the live range histogram
- // with a schedule as late as possible. Each fragment is live from
- // its earliest subload to its last WMMA consumer. The peak of the
+ // with a schedule as late as possible. Each fragment is live from
+ // its earliest subload to its last WMMA consumer. The peak of the
// histogram is the minimum VGPRs needed.
MapVector<Register, FragInfo> Frags;
for (LoadInfo &LI : Loads) {
>From bbefa159947f4c738ff4f2c892bfee4b0c2c13a5 Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Mon, 15 Jun 2026 19:25:58 +0000
Subject: [PATCH 10/20] LLVM_DEBUG messages and disable flag
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 144 +++++++++++++++++-
1 file changed, 139 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 4cdd0f5d1ecca..488826170a08b 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -31,6 +31,8 @@
#include "SIInstrInfo.h"
#include "llvm/CodeGen/ScheduleDAG.h"
#include "llvm/CodeGen/ScheduleDAGInstrs.h"
+#include "llvm/Support/CommandLine.h"
+#include "llvm/Support/Debug.h"
#include <optional>
#define DEBUG_TYPE "amdgpu-wmma-sched"
@@ -38,6 +40,12 @@ using namespace llvm;
namespace {
+// Disables the whole mutation (used to capture a no-mutation baseline schedule
+// for the before/after visualization).
+static cl::opt<bool> DisableWMMASchedule(
+ "amdgpu-wmma-sched-disable", cl::init(false),
+ cl::desc("Disable the AMDGPU WMMA ds_load scheduling mutation"));
+
// A single ds_load and their order among the WMMAs.
struct LoadInfo {
SUnit *SU;
@@ -73,7 +81,7 @@ class WMMASchedule : public ScheduleDAGMutation {
};
void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
- if (!ST.hasGFX1250Insts())
+ if (!ST.hasGFX1250Insts() || DisableWMMASchedule)
return;
const TargetSchedModel *SM = DAG->getSchedModel();
const SIInstrInfo *TII = ST.getInstrInfo();
@@ -112,18 +120,49 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (!LoadLatency || !LDSBandwidth || !WmmaLatency || Wmmas.empty())
return;
+ LLVM_DEBUG(
+ dbgs()
+ << "\n========================================================\n"
+ "AMDGPUWMMASchedule: ds_load scheduling mutation\n"
+ "Shapes where LDS (ds_load) prefetches sit relative to the WMMAs\n"
+ "that consume them, so the pre-RA scheduler doesn't bunch every\n"
+ "load at the top of the block (which stalls the WMMAs behind\n"
+ "s_wait_dscnt and inflates register pressure).\n"
+ "WMMAs are numbered W[0..N-1] in program order.\n"
+ "========================================================\n"
+ << "config: " << Wmmas.size() << " WMMAs, " << Loads.size()
+ << " ds_loads; loadlat=" << *LoadLatency << " wmmalat=" << *WmmaLatency
+ << " ldsbw=" << (unsigned)*LDSBandwidth << "\n");
+
// Order the WMMAs.
- auto [PrevSU, _] = *Wmmas.begin();
+ LLVM_DEBUG(
+ dbgs()
+ << "\n--- [1] WMMA ordering ------------------------------------\n"
+ "Chain each WMMA to the next in program order with an artificial\n"
+ "edge, so load placement can be reasoned about relative to fixed\n"
+ "W[] positions.\n");
+ auto [PrevSU, PrevPos] = *Wmmas.begin();
for (auto *It = std::next(Wmmas.begin()); It != Wmmas.end(); ++It) {
- auto [SU, _] = *It;
+ auto [SU, Pos] = *It;
DAG->addEdge(SU, SDep(PrevSU, SDep::Artificial));
+ LLVM_DEBUG(dbgs() << "[1] WMMA W[" << PrevPos << "] SU" << PrevSU->NodeNum
+ << " -> W[" << Pos << "] SU" << SU->NodeNum << "\n");
PrevSU = SU;
+ PrevPos = Pos;
}
// For each load, find earliest and latest consuming WMMA positions, and
// correct the ds_load -> earliest consumer data edge latency (Both the
// Succs and Preds SDep is updated)
+ LLVM_DEBUG(
+ dbgs()
+ << "\n--- [2] ds_load -> earliest consuming WMMA latency -------\n"
+ "For each ds_load, find which WMMAs consume it (MinPos = earliest,\n"
+ "MaxPos = latest). Update the data edge latency of the earliest\n"
+ "consumer to be the real LDS load latency, so the scheduler keeps\n"
+ "the load issued far enough ahead of the WMMAs that need it.\n");
for (LoadInfo &LI : Loads) {
+ SmallVector<SUnit *, 8> Consumers; // WMMA consumers (program order)
for (const SDep &D : LI.SU->Succs) {
if (D.getKind() != SDep::Data)
continue;
@@ -133,6 +172,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
continue;
LI.MinPos = std::min(LI.MinPos, It->second);
LI.MaxPos = std::max(LI.MaxPos, It->second);
+ Consumers.push_back(D.getSUnit());
}
if (LI.MinPos == UINT_MAX)
continue;
@@ -146,6 +186,15 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
P.setLatency(*LoadLatency);
EarliestConsumer->setDepthDirty();
LI.SU->setHeightDirty();
+ LLVM_DEBUG({
+ dbgs() << "[2] ds_load SU" << LI.SU->NodeNum << ": MinPos=" << LI.MinPos
+ << " MaxPos=" << LI.MaxPos << "; consumers W[" << LI.MinPos << ".."
+ << LI.MaxPos << "] (";
+ for (unsigned I = 0; I < Consumers.size(); ++I)
+ dbgs() << (I ? ", " : "") << "SU" << Consumers[I]->NodeNum;
+ dbgs() << "); set latency " << *LoadLatency << " on edge -> W["
+ << LI.MinPos << "]\n";
+ });
}
// Order the Loads
@@ -154,6 +203,13 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
});
// Chain consecutive loads with an LDS bandwidth latency
+ LLVM_DEBUG(
+ dbgs()
+ << "\n--- [3+4] ds_load -> ds_load spacing ---------------------\n"
+ "Sort loads by their earliest consumer, then chain consecutive\n"
+ "loads in that order with a small latency so they don't all issue\n"
+ "back to back and saturate the LDS bus (which would make the kernel\n"
+ "memory bound).\n");
SUnit *Prev = nullptr;
for (LoadInfo &LI : Loads) {
if (LI.MinPos == UINT_MAX)
@@ -162,6 +218,9 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
SDep D(Prev, SDep::Artificial);
D.setLatency(*LDSBandwidth);
DAG->addEdge(LI.SU, D);
+ LLVM_DEBUG(dbgs() << "[3+4] ds_load SU" << Prev->NodeNum << " -> SU"
+ << LI.SU->NodeNum << " (spacing latency "
+ << (unsigned)*LDSBandwidth << ")\n");
}
Prev = LI.SU;
}
@@ -171,17 +230,35 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
// if the ds_load -> ds_load edges requires spacing (ds_loads cannot be
// too close to each other or it could overwhelm the LDS bus and lead to
// the program being memory bound).
+ LLVM_DEBUG(
+ dbgs()
+ << "\n--- [lat]/[space] as late as possible cycle --------------\n"
+ "[lat]: latest cycle each load could issue and still feed its\n"
+ "earliest consumer in time (MinPos*wmmalat - loadlat).\n"
+ "[space]: loop through the ordered loads from last to first and\n"
+ "pull any that are too close to the next one earlier in order to\n"
+ "honor the ds_load -> ds_load spacing.\n");
for (LoadInfo &LI : Loads)
- if (LI.MinPos != UINT_MAX)
+ if (LI.MinPos != UINT_MAX) {
LI.LatestCycle = (long)LI.MinPos * (*WmmaLatency) - (long)(*LoadLatency);
+ LLVM_DEBUG(dbgs() << "[lat] ds_load SU" << LI.SU->NodeNum
+ << ": LatestCycle=" << LI.LatestCycle << " (W["
+ << LI.MinPos << "]*" << *WmmaLatency << " - "
+ << *LoadLatency << ")\n");
+ }
long PrevLatest = LONG_MAX;
for (int I = (int)Loads.size() - 1; I >= 0; --I) {
LoadInfo &LI = Loads[I];
if (LI.MinPos == UINT_MAX)
continue;
long Spaced = PrevLatest - (long)(*LDSBandwidth);
- if (Spaced < LI.LatestCycle)
+ if (Spaced < LI.LatestCycle) {
+ LLVM_DEBUG(dbgs() << "[space] ds_load SU" << LI.SU->NodeNum
+ << ": LatestCycle " << LI.LatestCycle << " -> " << Spaced
+ << " (spaced " << (unsigned)*LDSBandwidth
+ << " before next load's " << PrevLatest << ")\n");
LI.LatestCycle = Spaced;
+ }
PrevLatest = LI.LatestCycle;
}
@@ -189,6 +266,13 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
// with a schedule as late as possible. Each fragment is live from
// its earliest subload to its last WMMA consumer. The peak of the
// histogram is the minimum VGPRs needed.
+ LLVM_DEBUG(
+ dbgs()
+ << "\n--- [hist] live VGPR histogram (as late as possible schedule) ---\n"
+ "Group subloads that build one wide vreg into a 'fragment' (a\n"
+ "unit of VGPR pressure), then accumulate each fragment's VGPR\n"
+ "usage across the WMMA positions it is live over, under the\n"
+ "as late as possible schedule above.\n");
MapVector<Register, FragInfo> Frags;
for (LoadInfo &LI : Loads) {
if (LI.MinPos == UINT_MAX)
@@ -210,16 +294,44 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
unsigned StartPos = Pos < 0 ? 0 : (unsigned)Pos;
for (unsigned P = StartPos; P <= F.MaxPos && P < Wmmas.size(); ++P)
Hist[P] += F.VGPRs;
+ LLVM_DEBUG({
+ dbgs() << "[hist] frag (";
+ for (unsigned I = 0; I < F.Subloads.size(); ++I)
+ dbgs() << (I ? ", " : "") << "SU" << F.Subloads[I]->NodeNum;
+ dbgs() << ") (vgprs=" << F.VGPRs << ", LatestCycle=" << F.LatestCycle
+ << ") live over W[" << StartPos << ".." << F.MaxPos << "]\n";
+ });
}
unsigned Budget = 0;
for (unsigned P = 0; P < Wmmas.size(); ++P)
Budget = std::max(Budget, Hist[P]);
+ LLVM_DEBUG({
+ dbgs() << "\n--- [5] histogram BEFORE debunch (sets the budget) -------\n"
+ "Per WMMA position VGPR totals from the as late as possible\n"
+ "schedule. The peak becomes the VGPR budget: the debunch may\n"
+ "pull loads earlier as long as no position exceeds it.\n";
+ dbgs() << "[5] live-VGPR histogram BEFORE slack (min VGPRs / budget = "
+ << Budget << "):\n";
+ for (unsigned P = 0; P < Wmmas.size(); ++P)
+ if (Hist[P])
+ dbgs() << " W[" << P << "] = " << Hist[P] << "\n";
+ });
+
// For each fragment (in order), find the earliest position it
// can be placed so the live set never exceeds the budget, then
// add a WMMAS[Earliest] -> ds_load edge - this is what leads to
// the debunching.
+ LLVM_DEBUG(
+ dbgs()
+ << "\n--- [6] debunch: pull loads earlier into budget slack ----\n"
+ "For each fragment, scan earlier W[] positions while the budget\n"
+ "still has room. The earliest such position gets an artificial\n"
+ "WMMA -> ds_load edge that stops the scheduler from bunching that load\n"
+ "any earlier. 'unconstrained' = it already fits at W[0], so no edge\n"
+ "is needed; the histogram is updated cumulatively so later\n"
+ "fragments only use the slack that's left over.\n");
for (auto &KV : Frags) {
FragInfo &F = KV.second;
long Pos = F.LatestCycle / (long)(*WmmaLatency);
@@ -235,12 +347,34 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
// exceed the budget.
for (unsigned P = Earliest; P < LateStartPos; ++P)
Hist[P] += F.VGPRs;
+ LLVM_DEBUG({
+ dbgs() << "[6] frag (";
+ for (unsigned I = 0; I < F.Subloads.size(); ++I)
+ dbgs() << (I ? ", " : "") << "SU" << F.Subloads[I]->NodeNum;
+ dbgs() << ") (vgprs=" << F.VGPRs << ", consumers W[" << F.MinPos << ".."
+ << F.MaxPos << "]) earliest=W[" << Earliest << "]"
+ << (Earliest ? " edge added\n" : " unconstrained\n");
+ });
// No need to add an edge if the load can be scheduled at the beginning.
if (Earliest == 0)
continue;
for (SUnit *L : F.Subloads)
DAG->addEdge(L, SDep(Wmmas.begin()[Earliest].first, SDep::Artificial));
}
+
+ LLVM_DEBUG({
+ unsigned Peak = 0;
+ for (unsigned P = 0; P < Wmmas.size(); ++P)
+ Peak = std::max(Peak, Hist[P]);
+ dbgs() << "\n--- [6] histogram AFTER debunch -------------------------\n"
+ "Same usage after the debunch edges. Loads now sit as\n"
+ "early as the budget allows; the peak should still be within\n"
+ "the budget from [5].\n";
+ dbgs() << "[6] live-VGPR histogram AFTER slack (peak = " << Peak << "):\n";
+ for (unsigned P = 0; P < Wmmas.size(); ++P)
+ if (Hist[P])
+ dbgs() << " W[" << P << "] = " << Hist[P] << "\n";
+ });
}
} // end namespace
>From 83967fa8db83f0245ecbf98f72d5e6bae2349522 Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Mon, 22 Jun 2026 23:23:47 +0000
Subject: [PATCH 11/20] WIP: Flag for ds_load latencies
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 28 +++++++++++++++++++
1 file changed, 28 insertions(+)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 488826170a08b..5bdc32a5b3bd0 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -33,6 +33,7 @@
#include "llvm/CodeGen/ScheduleDAGInstrs.h"
#include "llvm/Support/CommandLine.h"
#include "llvm/Support/Debug.h"
+#include <cstdlib>
#include <optional>
#define DEBUG_TYPE "amdgpu-wmma-sched"
@@ -46,6 +47,17 @@ static cl::opt<bool> DisableWMMASchedule(
"amdgpu-wmma-sched-disable", cl::init(false),
cl::desc("Disable the AMDGPU WMMA ds_load scheduling mutation"));
+// Overrides the ds_load (LDS) latency the mutation uses to space loads ahead of
+// their consuming WMMAs. 0 (default, the unset state) uses the scheduling
+// model's computeInstrLatency value (20 on gfx1250). Set explicitly to sweep
+// how aggressively loads are hoisted, e.g. -amdgpu-wmma-sched-loadlat=64. A
+// value of 0 *is* honored when passed explicitly (getNumOccurrences()), letting
+// the sweep include latency=0.
+static cl::opt<unsigned> WMMALoadLatency(
+ "amdgpu-wmma-sched-loadlat", cl::init(0),
+ cl::desc("Override the ds_load latency (cycles) used by the AMDGPU WMMA "
+ "ds_load scheduling mutation; 0/unset = use the sched model"));
+
// A single ds_load and their order among the WMMAs.
struct LoadInfo {
SUnit *SU;
@@ -83,6 +95,13 @@ class WMMASchedule : public ScheduleDAGMutation {
void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (!ST.hasGFX1250Insts() || DisableWMMASchedule)
return;
+ // Env override for the Triton path: Triton applies the `flags` it passes to
+ // the backend as *bool, name-only* cl options (python/src/llvm.cc), so a
+ // valued option can't be set that way. Reading the environment here (the
+ // backend runs in-process) lets the sweep harness disable the mutation or
+ // override the ds_load latency per-run. The cl::opts above stay for llc/opt.
+ if (const char *E = std::getenv("AMDGPU_WMMA_SCHED_DISABLE"); E && std::atoi(E))
+ return;
const TargetSchedModel *SM = DAG->getSchedModel();
const SIInstrInfo *TII = ST.getInstrInfo();
@@ -120,6 +139,15 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (!LoadLatency || !LDSBandwidth || !WmmaLatency || Wmmas.empty())
return;
+ // Allow the ds_load latency to be overridden so we can sweep how aggressively
+ // loads are hoisted ahead of their WMMAs (the model default is 20). The env
+ // var wins (the Triton sweep path); otherwise the cl::opt, gated on
+ // getNumOccurrences() (not value) so an explicit ...loadlat=0 is honored.
+ if (const char *E = std::getenv("AMDGPU_WMMA_SCHED_LOADLAT"); E && *E)
+ LoadLatency = (unsigned)std::atoi(E);
+ else if (WMMALoadLatency.getNumOccurrences())
+ LoadLatency = WMMALoadLatency;
+
LLVM_DEBUG(
dbgs()
<< "\n========================================================\n"
>From f8a382ca8a82ba6935fad94cb1d6501879938f87 Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Tue, 28 Jul 2026 02:07:12 +0000
Subject: [PATCH 12/20] Register WMMA ds_load mutation under CoExec scheduler
too
---
llvm/lib/Target/AMDGPU/AMDGPUCoExecSchedStrategy.cpp | 3 +++
1 file changed, 3 insertions(+)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUCoExecSchedStrategy.cpp b/llvm/lib/Target/AMDGPU/AMDGPUCoExecSchedStrategy.cpp
index bee2093cdc12c..610295011da01 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUCoExecSchedStrategy.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUCoExecSchedStrategy.cpp
@@ -13,6 +13,7 @@
#include "AMDGPUCoExecSchedStrategy.h"
#include "AMDGPUIGroupLP.h"
+#include "AMDGPUWMMASchedule.h"
#include "llvm/Support/Debug.h"
using namespace llvm;
@@ -712,6 +713,8 @@ llvm::createGCNCoExecMachineScheduler(MachineSchedContext *C) {
ScheduleDAGMILive *DAG = new GCNScheduleDAGMILive(
C, std::make_unique<AMDGPUCoExecSchedStrategy>(C));
DAG->addMutation(createIGroupLPDAGMutation(AMDGPU::SchedulingPhase::Initial));
+ // Also run the WMMA ds_load scheduling mutation under CoExec.
+ DAG->addMutation(createAMDGPUWMMAScheduleDAGMutation(C->MF));
return DAG;
}
>From fa12ac38372ff0b17cc79c206c66d2602a7cfd5a Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Tue, 28 Jul 2026 02:18:44 +0000
Subject: [PATCH 13/20] [AMDGPU] Remove experimental flags from WMMA ds_load
scheduling mutation
Drops the amdgpu-wmma-sched-* cl::opts and AMDGPU_WMMA_SCHED_* env overrides; the
ds_load latency now comes from the scheduling model. Registers the mutation in the
default GCN scheduler.
---
.../AMDGPU/AMDGPUCoExecSchedStrategy.cpp | 1 -
.../lib/Target/AMDGPU/AMDGPUTargetMachine.cpp | 209 +++++++++---------
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 36 +--
3 files changed, 110 insertions(+), 136 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUCoExecSchedStrategy.cpp b/llvm/lib/Target/AMDGPU/AMDGPUCoExecSchedStrategy.cpp
index 610295011da01..d021db8f7fc25 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUCoExecSchedStrategy.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUCoExecSchedStrategy.cpp
@@ -713,7 +713,6 @@ llvm::createGCNCoExecMachineScheduler(MachineSchedContext *C) {
ScheduleDAGMILive *DAG = new GCNScheduleDAGMILive(
C, std::make_unique<AMDGPUCoExecSchedStrategy>(C));
DAG->addMutation(createIGroupLPDAGMutation(AMDGPU::SchedulingPhase::Initial));
- // Also run the WMMA ds_load scheduling mutation under CoExec.
DAG->addMutation(createAMDGPUWMMAScheduleDAGMutation(C->MF));
return DAG;
}
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
index 6236959bd136a..48673fc0d2ad1 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUTargetMachine.cpp
@@ -185,13 +185,13 @@ class AMDGPUCodeGenPassBuilder
class SGPRRegisterRegAlloc : public RegisterRegAllocBase<SGPRRegisterRegAlloc> {
public:
SGPRRegisterRegAlloc(const char *N, const char *D, FunctionPassCtor C)
- : RegisterRegAllocBase(N, D, C) {}
+ : RegisterRegAllocBase(N, D, C) {}
};
class VGPRRegisterRegAlloc : public RegisterRegAllocBase<VGPRRegisterRegAlloc> {
public:
VGPRRegisterRegAlloc(const char *N, const char *D, FunctionPassCtor C)
- : RegisterRegAllocBase(N, D, C) {}
+ : RegisterRegAllocBase(N, D, C) {}
};
class WWMRegisterRegAlloc : public RegisterRegAllocBase<WWMRegisterRegAlloc> {
@@ -234,21 +234,19 @@ static llvm::once_flag InitializeDefaultVGPRRegisterAllocatorFlag;
static llvm::once_flag InitializeDefaultWWMRegisterAllocatorFlag;
static SGPRRegisterRegAlloc
- defaultSGPRRegAlloc("default",
- "pick SGPR register allocator based on -O option",
- useDefaultRegisterAllocator);
+defaultSGPRRegAlloc("default",
+ "pick SGPR register allocator based on -O option",
+ useDefaultRegisterAllocator);
static cl::opt<SGPRRegisterRegAlloc::FunctionPassCtor, false,
RegisterPassParser<SGPRRegisterRegAlloc>>
- SGPRRegAlloc("sgpr-regalloc", cl::Hidden,
- cl::init(&useDefaultRegisterAllocator),
- cl::desc("Register allocator to use for SGPRs"));
+SGPRRegAlloc("sgpr-regalloc", cl::Hidden, cl::init(&useDefaultRegisterAllocator),
+ cl::desc("Register allocator to use for SGPRs"));
static cl::opt<VGPRRegisterRegAlloc::FunctionPassCtor, false,
RegisterPassParser<VGPRRegisterRegAlloc>>
- VGPRRegAlloc("vgpr-regalloc", cl::Hidden,
- cl::init(&useDefaultRegisterAllocator),
- cl::desc("Register allocator to use for VGPRs"));
+VGPRRegAlloc("vgpr-regalloc", cl::Hidden, cl::init(&useDefaultRegisterAllocator),
+ cl::desc("Register allocator to use for VGPRs"));
static cl::opt<WWMRegisterRegAlloc::FunctionPassCtor, false,
RegisterPassParser<WWMRegisterRegAlloc>>
@@ -376,25 +374,22 @@ static FunctionPass *createFastWWMRegisterAllocator() {
return createFastRegisterAllocator(onlyAllocateWWMRegs, false);
}
-static SGPRRegisterRegAlloc basicRegAllocSGPR("basic",
- "basic register allocator",
- createBasicSGPRRegisterAllocator);
-static SGPRRegisterRegAlloc
- greedyRegAllocSGPR("greedy", "greedy register allocator",
- createGreedySGPRRegisterAllocator);
+static SGPRRegisterRegAlloc basicRegAllocSGPR(
+ "basic", "basic register allocator", createBasicSGPRRegisterAllocator);
+static SGPRRegisterRegAlloc greedyRegAllocSGPR(
+ "greedy", "greedy register allocator", createGreedySGPRRegisterAllocator);
-static SGPRRegisterRegAlloc fastRegAllocSGPR("fast", "fast register allocator",
- createFastSGPRRegisterAllocator);
+static SGPRRegisterRegAlloc fastRegAllocSGPR(
+ "fast", "fast register allocator", createFastSGPRRegisterAllocator);
-static VGPRRegisterRegAlloc basicRegAllocVGPR("basic",
- "basic register allocator",
- createBasicVGPRRegisterAllocator);
-static VGPRRegisterRegAlloc
- greedyRegAllocVGPR("greedy", "greedy register allocator",
- createGreedyVGPRRegisterAllocator);
-static VGPRRegisterRegAlloc fastRegAllocVGPR("fast", "fast register allocator",
- createFastVGPRRegisterAllocator);
+static VGPRRegisterRegAlloc basicRegAllocVGPR(
+ "basic", "basic register allocator", createBasicVGPRRegisterAllocator);
+static VGPRRegisterRegAlloc greedyRegAllocVGPR(
+ "greedy", "greedy register allocator", createGreedyVGPRRegisterAllocator);
+
+static VGPRRegisterRegAlloc fastRegAllocVGPR(
+ "fast", "fast register allocator", createFastVGPRRegisterAllocator);
static WWMRegisterRegAlloc basicRegAllocWWMReg("basic",
"basic register allocator",
createBasicWWMRegisterAllocator);
@@ -411,14 +406,14 @@ static bool isLTOPreLink(ThinOrFullLTOPhase Phase) {
} // anonymous namespace
static cl::opt<bool>
- EnableEarlyIfConversion("amdgpu-early-ifcvt", cl::Hidden,
- cl::desc("Run early if-conversion"),
- cl::init(false));
+EnableEarlyIfConversion("amdgpu-early-ifcvt", cl::Hidden,
+ cl::desc("Run early if-conversion"),
+ cl::init(false));
static cl::opt<bool>
- OptExecMaskPreRA("amdgpu-opt-exec-mask-pre-ra", cl::Hidden,
- cl::desc("Run pre-RA exec mask optimizations"),
- cl::init(true));
+OptExecMaskPreRA("amdgpu-opt-exec-mask-pre-ra", cl::Hidden,
+ cl::desc("Run pre-RA exec mask optimizations"),
+ cl::init(true));
static cl::opt<bool>
LowerCtorDtor("amdgpu-lower-global-ctor-dtor",
@@ -426,27 +421,32 @@ static cl::opt<bool>
cl::init(true), cl::Hidden);
// Option to disable vectorizer for tests.
-static cl::opt<bool>
- EnableLoadStoreVectorizer("amdgpu-load-store-vectorizer",
- cl::desc("Enable load store vectorizer"),
- cl::init(true), cl::Hidden);
+static cl::opt<bool> EnableLoadStoreVectorizer(
+ "amdgpu-load-store-vectorizer",
+ cl::desc("Enable load store vectorizer"),
+ cl::init(true),
+ cl::Hidden);
// Option to control global loads scalarization
-static cl::opt<bool>
- ScalarizeGlobal("amdgpu-scalarize-global-loads",
- cl::desc("Enable global load scalarization"),
- cl::init(true), cl::Hidden);
+static cl::opt<bool> ScalarizeGlobal(
+ "amdgpu-scalarize-global-loads",
+ cl::desc("Enable global load scalarization"),
+ cl::init(true),
+ cl::Hidden);
// Option to run internalize pass.
static cl::opt<bool> InternalizeSymbols(
- "amdgpu-internalize-symbols",
- cl::desc("Enable elimination of non-kernel functions and unused globals"),
- cl::init(false), cl::Hidden);
+ "amdgpu-internalize-symbols",
+ cl::desc("Enable elimination of non-kernel functions and unused globals"),
+ cl::init(false),
+ cl::Hidden);
// Option to inline all early.
-static cl::opt<bool> EarlyInlineAll("amdgpu-early-inline-all",
- cl::desc("Inline all functions early"),
- cl::init(false), cl::Hidden);
+static cl::opt<bool> EarlyInlineAll(
+ "amdgpu-early-inline-all",
+ cl::desc("Inline all functions early"),
+ cl::init(false),
+ cl::Hidden);
static cl::opt<bool> RemoveIncompatibleFunctions(
"amdgpu-enable-remove-incompatible-functions", cl::Hidden,
@@ -454,35 +454,39 @@ static cl::opt<bool> RemoveIncompatibleFunctions(
"use features not supported by the target GPU"),
cl::init(true));
-static cl::opt<bool> EnableSDWAPeephole("amdgpu-sdwa-peephole",
- cl::desc("Enable SDWA peepholer"),
- cl::init(true));
+static cl::opt<bool> EnableSDWAPeephole(
+ "amdgpu-sdwa-peephole",
+ cl::desc("Enable SDWA peepholer"),
+ cl::init(true));
-static cl::opt<bool> EnableDPPCombine("amdgpu-dpp-combine",
- cl::desc("Enable DPP combiner"),
- cl::init(true));
+static cl::opt<bool> EnableDPPCombine(
+ "amdgpu-dpp-combine",
+ cl::desc("Enable DPP combiner"),
+ cl::init(true));
// Enable address space based alias analysis
-static cl::opt<bool>
- EnableAMDGPUAliasAnalysis("enable-amdgpu-aa", cl::Hidden,
- cl::desc("Enable AMDGPU Alias Analysis"),
- cl::init(true));
+static cl::opt<bool> EnableAMDGPUAliasAnalysis("enable-amdgpu-aa", cl::Hidden,
+ cl::desc("Enable AMDGPU Alias Analysis"),
+ cl::init(true));
// Enable lib calls simplifications
-static cl::opt<bool>
- EnableLibCallSimplify("amdgpu-simplify-libcall",
- cl::desc("Enable amdgpu library simplifications"),
- cl::init(true), cl::Hidden);
+static cl::opt<bool> EnableLibCallSimplify(
+ "amdgpu-simplify-libcall",
+ cl::desc("Enable amdgpu library simplifications"),
+ cl::init(true),
+ cl::Hidden);
static cl::opt<bool> EnableLowerKernelArguments(
- "amdgpu-ir-lower-kernel-arguments",
- cl::desc("Lower kernel argument loads in IR pass"), cl::init(true),
- cl::Hidden);
+ "amdgpu-ir-lower-kernel-arguments",
+ cl::desc("Lower kernel argument loads in IR pass"),
+ cl::init(true),
+ cl::Hidden);
static cl::opt<bool> EnableRegReassign(
- "amdgpu-reassign-regs",
- cl::desc("Enable register reassign optimizations on gfx10+"),
- cl::init(true), cl::Hidden);
+ "amdgpu-reassign-regs",
+ cl::desc("Enable register reassign optimizations on gfx10+"),
+ cl::init(true),
+ cl::Hidden);
static cl::opt<bool> OptVGPRLiveRange(
"amdgpu-opt-vgpr-liverange",
@@ -500,10 +504,11 @@ static cl::opt<ScanOptions> AMDGPUAtomicOptimizerStrategy(
clEnumValN(ScanOptions::None, "None", "Disable atomic optimizer")));
// Enable Mode register optimization
-static cl::opt<bool>
- EnableSIModeRegisterPass("amdgpu-mode-register",
- cl::desc("Enable mode register pass"),
- cl::init(true), cl::Hidden);
+static cl::opt<bool> EnableSIModeRegisterPass(
+ "amdgpu-mode-register",
+ cl::desc("Enable mode register pass"),
+ cl::init(true),
+ cl::Hidden);
// Enable GFX11+ s_delay_alu insertion
static cl::opt<bool>
@@ -519,16 +524,19 @@ static cl::opt<bool>
// Option is used in lit tests to prevent deadcoding of patterns inspected.
static cl::opt<bool>
- EnableDCEInRA("amdgpu-dce-in-ra", cl::init(true), cl::Hidden,
- cl::desc("Enable machine DCE inside regalloc"));
+EnableDCEInRA("amdgpu-dce-in-ra",
+ cl::init(true), cl::Hidden,
+ cl::desc("Enable machine DCE inside regalloc"));
static cl::opt<bool> EnableSetWavePriority("amdgpu-set-wave-priority",
cl::desc("Adjust wave priority"),
cl::init(false), cl::Hidden);
-static cl::opt<bool> EnableScalarIRPasses("amdgpu-scalar-ir-passes",
- cl::desc("Enable scalar IR passes"),
- cl::init(true), cl::Hidden);
+static cl::opt<bool> EnableScalarIRPasses(
+ "amdgpu-scalar-ir-passes",
+ cl::desc("Enable scalar IR passes"),
+ cl::init(true),
+ cl::Hidden);
static cl::opt<bool> EnableLowerExecSync(
"amdgpu-enable-lower-exec-sync",
@@ -552,10 +560,10 @@ static cl::opt<bool, true> EnableLowerModuleLDS(
cl::location(AMDGPUTargetMachine::EnableLowerModuleLDS), cl::init(true),
cl::Hidden);
-static cl::opt<bool>
- EnablePreRAOptimizations("amdgpu-enable-pre-ra-optimizations",
- cl::desc("Enable Pre-RA optimizations pass"),
- cl::init(true), cl::Hidden);
+static cl::opt<bool> EnablePreRAOptimizations(
+ "amdgpu-enable-pre-ra-optimizations",
+ cl::desc("Enable Pre-RA optimizations pass"), cl::init(true),
+ cl::Hidden);
static cl::opt<bool> EnablePromoteKernelArguments(
"amdgpu-enable-promote-kernel-arguments",
@@ -614,10 +622,10 @@ static cl::opt<bool> EnableRewritePartialRegUses(
cl::desc("Enable rewrite partial reg uses pass"), cl::init(true),
cl::Hidden);
-static cl::opt<bool>
- EnableHipStdPar("amdgpu-enable-hipstdpar",
- cl::desc("Enable HIP Standard Parallelism Offload support"),
- cl::init(false), cl::Hidden);
+static cl::opt<bool> EnableHipStdPar(
+ "amdgpu-enable-hipstdpar",
+ cl::desc("Enable HIP Standard Parallelism Offload support"), cl::init(false),
+ cl::Hidden);
static cl::opt<bool>
EnableAMDGPUAttributor("amdgpu-attributor-enable",
@@ -743,8 +751,8 @@ static ScheduleDAGInstrs *createSIMachineScheduler(MachineSchedContext *C) {
static ScheduleDAGInstrs *
createGCNMaxOccupancyMachineScheduler(MachineSchedContext *C) {
const GCNSubtarget &ST = C->MF->getSubtarget<GCNSubtarget>();
- ScheduleDAGMILive *DAG = new GCNScheduleDAGMILive(
- C, std::make_unique<GCNMaxOccupancySchedStrategy>(C));
+ ScheduleDAGMILive *DAG =
+ new GCNScheduleDAGMILive(C, std::make_unique<GCNMaxOccupancySchedStrategy>(C));
DAG->addMutation(createLoadClusterDAGMutation(DAG->TII, DAG->TRI));
if (ST.shouldClusterStores())
DAG->addMutation(createStoreClusterDAGMutation(DAG->TII, DAG->TRI));
@@ -810,13 +818,14 @@ createIterativeILPMachineScheduler(MachineSchedContext *C) {
return DAG;
}
-static MachineSchedRegistry SISchedRegistry("si", "Run SI's custom scheduler",
- createSIMachineScheduler);
+static MachineSchedRegistry
+SISchedRegistry("si", "Run SI's custom scheduler",
+ createSIMachineScheduler);
static MachineSchedRegistry
- GCNMaxOccupancySchedRegistry("gcn-max-occupancy",
- "Run GCN scheduler to maximize occupancy",
- createGCNMaxOccupancyMachineScheduler);
+GCNMaxOccupancySchedRegistry("gcn-max-occupancy",
+ "Run GCN scheduler to maximize occupancy",
+ createGCNMaxOccupancyMachineScheduler);
static MachineSchedRegistry
GCNMaxILPSchedRegistry("gcn-max-ilp", "Run GCN scheduler to maximize ilp",
@@ -1511,7 +1520,7 @@ void AMDGPUPassConfig::addIRPasses() {
AAResults &AAR) {
if (auto *WrapperPass = P.getAnalysisIfAvailable<AMDGPUAAWrapperPass>())
AAR.addAAResult(WrapperPass->getResult());
- }));
+ }));
}
if (TM.getTargetTriple().isAMDGCN()) {
@@ -2125,8 +2134,8 @@ bool GCNTargetMachine::parseMachineFunctionInfo(
AMDGPU::SGPR_32RegClass,
MFI->ArgInfo.PrivateSegmentSize, 0, 0) ||
parseAndCheckArgument(YamlMFI.ArgInfo->LDSKernelId,
- AMDGPU::SGPR_32RegClass, MFI->ArgInfo.LDSKernelId,
- 0, 1) ||
+ AMDGPU::SGPR_32RegClass,
+ MFI->ArgInfo.LDSKernelId, 0, 1) ||
parseAndCheckArgument(YamlMFI.ArgInfo->WorkGroupIDX,
AMDGPU::SGPR_32RegClass, MFI->ArgInfo.WorkGroupIDX,
0, 1) ||
@@ -2149,14 +2158,14 @@ bool GCNTargetMachine::parseMachineFunctionInfo(
AMDGPU::SReg_64RegClass,
MFI->ArgInfo.ImplicitBufferPtr, 2, 0) ||
parseAndCheckArgument(YamlMFI.ArgInfo->WorkItemIDX,
- AMDGPU::VGPR_32RegClass, MFI->ArgInfo.WorkItemIDX,
- 0, 0) ||
+ AMDGPU::VGPR_32RegClass,
+ MFI->ArgInfo.WorkItemIDX, 0, 0) ||
parseAndCheckArgument(YamlMFI.ArgInfo->WorkItemIDY,
- AMDGPU::VGPR_32RegClass, MFI->ArgInfo.WorkItemIDY,
- 0, 0) ||
+ AMDGPU::VGPR_32RegClass,
+ MFI->ArgInfo.WorkItemIDY, 0, 0) ||
parseAndCheckArgument(YamlMFI.ArgInfo->WorkItemIDZ,
- AMDGPU::VGPR_32RegClass, MFI->ArgInfo.WorkItemIDZ,
- 0, 0)))
+ AMDGPU::VGPR_32RegClass,
+ MFI->ArgInfo.WorkItemIDZ, 0, 0)))
return true;
// Parse FirstKernArgPreloadReg separately, since it's a Register,
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 5bdc32a5b3bd0..8b22a284d5fda 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -33,7 +33,6 @@
#include "llvm/CodeGen/ScheduleDAGInstrs.h"
#include "llvm/Support/CommandLine.h"
#include "llvm/Support/Debug.h"
-#include <cstdlib>
#include <optional>
#define DEBUG_TYPE "amdgpu-wmma-sched"
@@ -41,23 +40,6 @@ using namespace llvm;
namespace {
-// Disables the whole mutation (used to capture a no-mutation baseline schedule
-// for the before/after visualization).
-static cl::opt<bool> DisableWMMASchedule(
- "amdgpu-wmma-sched-disable", cl::init(false),
- cl::desc("Disable the AMDGPU WMMA ds_load scheduling mutation"));
-
-// Overrides the ds_load (LDS) latency the mutation uses to space loads ahead of
-// their consuming WMMAs. 0 (default, the unset state) uses the scheduling
-// model's computeInstrLatency value (20 on gfx1250). Set explicitly to sweep
-// how aggressively loads are hoisted, e.g. -amdgpu-wmma-sched-loadlat=64. A
-// value of 0 *is* honored when passed explicitly (getNumOccurrences()), letting
-// the sweep include latency=0.
-static cl::opt<unsigned> WMMALoadLatency(
- "amdgpu-wmma-sched-loadlat", cl::init(0),
- cl::desc("Override the ds_load latency (cycles) used by the AMDGPU WMMA "
- "ds_load scheduling mutation; 0/unset = use the sched model"));
-
// A single ds_load and their order among the WMMAs.
struct LoadInfo {
SUnit *SU;
@@ -93,14 +75,7 @@ class WMMASchedule : public ScheduleDAGMutation {
};
void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
- if (!ST.hasGFX1250Insts() || DisableWMMASchedule)
- return;
- // Env override for the Triton path: Triton applies the `flags` it passes to
- // the backend as *bool, name-only* cl options (python/src/llvm.cc), so a
- // valued option can't be set that way. Reading the environment here (the
- // backend runs in-process) lets the sweep harness disable the mutation or
- // override the ds_load latency per-run. The cl::opts above stay for llc/opt.
- if (const char *E = std::getenv("AMDGPU_WMMA_SCHED_DISABLE"); E && std::atoi(E))
+ if (!ST.hasGFX1250Insts())
return;
const TargetSchedModel *SM = DAG->getSchedModel();
const SIInstrInfo *TII = ST.getInstrInfo();
@@ -139,15 +114,6 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (!LoadLatency || !LDSBandwidth || !WmmaLatency || Wmmas.empty())
return;
- // Allow the ds_load latency to be overridden so we can sweep how aggressively
- // loads are hoisted ahead of their WMMAs (the model default is 20). The env
- // var wins (the Triton sweep path); otherwise the cl::opt, gated on
- // getNumOccurrences() (not value) so an explicit ...loadlat=0 is honored.
- if (const char *E = std::getenv("AMDGPU_WMMA_SCHED_LOADLAT"); E && *E)
- LoadLatency = (unsigned)std::atoi(E);
- else if (WMMALoadLatency.getNumOccurrences())
- LoadLatency = WMMALoadLatency;
-
LLVM_DEBUG(
dbgs()
<< "\n========================================================\n"
>From 39635919c2db88efc40fba455996530bf6c6b1e6 Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Wed, 9 Sep 2026 16:10:28 -0600
Subject: [PATCH 14/20] clang-format
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 17 +++++++++--------
1 file changed, 9 insertions(+), 8 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 8b22a284d5fda..9daabe796c5cd 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -133,7 +133,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
dbgs()
<< "\n--- [1] WMMA ordering ------------------------------------\n"
"Chain each WMMA to the next in program order with an artificial\n"
- "edge, so load placement can be reasoned about relative to fixed\n"
+ "edge, so load placement can be reasoned about relative to fixed\n"
"W[] positions.\n");
auto [PrevSU, PrevPos] = *Wmmas.begin();
for (auto *It = std::next(Wmmas.begin()); It != Wmmas.end(); ++It) {
@@ -201,8 +201,8 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
dbgs()
<< "\n--- [3+4] ds_load -> ds_load spacing ---------------------\n"
"Sort loads by their earliest consumer, then chain consecutive\n"
- "loads in that order with a small latency so they don't all issue\n"
- "back to back and saturate the LDS bus (which would make the kernel\n"
+ "loads in that order with a small latency so they don't all issue\n"
+ "back to back and saturate the LDS bus (which would make the kernel\n"
"memory bound).\n");
SUnit *Prev = nullptr;
for (LoadInfo &LI : Loads) {
@@ -230,7 +230,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
"[lat]: latest cycle each load could issue and still feed its\n"
"earliest consumer in time (MinPos*wmmalat - loadlat).\n"
"[space]: loop through the ordered loads from last to first and\n"
- "pull any that are too close to the next one earlier in order to\n"
+ "pull any that are too close to the next one earlier in order to\n"
"honor the ds_load -> ds_load spacing.\n");
for (LoadInfo &LI : Loads)
if (LI.MinPos != UINT_MAX) {
@@ -248,8 +248,8 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
long Spaced = PrevLatest - (long)(*LDSBandwidth);
if (Spaced < LI.LatestCycle) {
LLVM_DEBUG(dbgs() << "[space] ds_load SU" << LI.SU->NodeNum
- << ": LatestCycle " << LI.LatestCycle << " -> " << Spaced
- << " (spaced " << (unsigned)*LDSBandwidth
+ << ": LatestCycle " << LI.LatestCycle << " -> "
+ << Spaced << " (spaced " << (unsigned)*LDSBandwidth
<< " before next load's " << PrevLatest << ")\n");
LI.LatestCycle = Spaced;
}
@@ -303,7 +303,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
LLVM_DEBUG({
dbgs() << "\n--- [5] histogram BEFORE debunch (sets the budget) -------\n"
- "Per WMMA position VGPR totals from the as late as possible\n"
+ "Per WMMA position VGPR totals from the as late as possible\n"
"schedule. The peak becomes the VGPR budget: the debunch may\n"
"pull loads earlier as long as no position exceeds it.\n";
dbgs() << "[5] live-VGPR histogram BEFORE slack (min VGPRs / budget = "
@@ -322,7 +322,8 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
<< "\n--- [6] debunch: pull loads earlier into budget slack ----\n"
"For each fragment, scan earlier W[] positions while the budget\n"
"still has room. The earliest such position gets an artificial\n"
- "WMMA -> ds_load edge that stops the scheduler from bunching that load\n"
+ "WMMA -> ds_load edge that stops the scheduler from bunching that "
+ "load\n"
"any earlier. 'unconstrained' = it already fits at W[0], so no edge\n"
"is needed; the histogram is updated cumulatively so later\n"
"fragments only use the slack that's left over.\n");
>From 8c0e2a96406def5176afad529d3fffde54fc898b Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Wed, 9 Sep 2026 17:47:33 -0600
Subject: [PATCH 15/20] hasInstrSchedModel check
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 2 ++
1 file changed, 2 insertions(+)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 9daabe796c5cd..2838ca85ee0f2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -78,6 +78,8 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (!ST.hasGFX1250Insts())
return;
const TargetSchedModel *SM = DAG->getSchedModel();
+ if (!SM->hasInstrSchedModel())
+ return;
const SIInstrInfo *TII = ST.getInstrInfo();
// Gather WMMAs (numbered in program order) and ds_loads.
>From 01386ae8867074bfc4f4e5fdacf8fe7669230876 Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Wed, 9 Sep 2026 18:26:54 -0600
Subject: [PATCH 16/20] Rather than one global load latency and reciprocal
throughput, each individual load's latency and receiprocal throughput is used
instead.
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 41 +++++++++----------
1 file changed, 20 insertions(+), 21 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 2838ca85ee0f2..84acbc3624ab2 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -85,9 +85,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
// Gather WMMAs (numbered in program order) and ds_loads.
MapVector<SUnit *, unsigned> Wmmas; // Ordered WMMA SUnits
SmallVector<LoadInfo> Loads;
- std::optional<unsigned> LoadLatency;
std::optional<unsigned> WmmaLatency;
- std::optional<double> LDSBandwidth;
for (SUnit &SU : DAG->SUnits) {
MachineInstr *MI = SU.getInstr();
@@ -103,17 +101,12 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
}
// Gather DS_LOADs
- if (TII->isDS(*MI) && MI->mayLoad()) {
- if (!LoadLatency)
- LoadLatency = SM->computeInstrLatency(MI);
- if (!LDSBandwidth)
- LDSBandwidth = std::ceil(SM->computeReciprocalThroughput(MI));
+ if (TII->isDS(*MI) && MI->mayLoad())
Loads.push_back({&SU});
- }
}
// The following means the DAG Mutation cannot do anything useful.
- if (!LoadLatency || !LDSBandwidth || !WmmaLatency || Wmmas.empty())
+ if (Loads.empty() || !WmmaLatency || Wmmas.empty())
return;
LLVM_DEBUG(
@@ -127,8 +120,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
"WMMAs are numbered W[0..N-1] in program order.\n"
"========================================================\n"
<< "config: " << Wmmas.size() << " WMMAs, " << Loads.size()
- << " ds_loads; loadlat=" << *LoadLatency << " wmmalat=" << *WmmaLatency
- << " ldsbw=" << (unsigned)*LDSBandwidth << "\n");
+ << " ds_loads; wmmalat=" << *WmmaLatency << "\n");
// Order the WMMAs.
LLVM_DEBUG(
@@ -173,13 +165,14 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (LI.MinPos == UINT_MAX)
continue;
SUnit *EarliestConsumer = Wmmas.begin()[LI.MinPos].first;
+ const unsigned LoadLatency = SM->computeInstrLatency(LI.SU->getInstr());
// Correct latency of edges between ds_load and earliest WMMA consumer
for (SDep &S : LI.SU->Succs)
if (S.getSUnit() == EarliestConsumer && S.getKind() == SDep::Data)
- S.setLatency(*LoadLatency);
+ S.setLatency(LoadLatency);
for (SDep &P : EarliestConsumer->Preds)
if (P.getSUnit() == LI.SU && P.getKind() == SDep::Data)
- P.setLatency(*LoadLatency);
+ P.setLatency(LoadLatency);
EarliestConsumer->setDepthDirty();
LI.SU->setHeightDirty();
LLVM_DEBUG({
@@ -188,7 +181,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
<< LI.MaxPos << "] (";
for (unsigned I = 0; I < Consumers.size(); ++I)
dbgs() << (I ? ", " : "") << "SU" << Consumers[I]->NodeNum;
- dbgs() << "); set latency " << *LoadLatency << " on edge -> W["
+ dbgs() << "); set latency " << LoadLatency << " on edge -> W["
<< LI.MinPos << "]\n";
});
}
@@ -211,12 +204,14 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (LI.MinPos == UINT_MAX)
continue;
if (Prev) {
+ const unsigned Spacing = static_cast<unsigned>(
+ std::ceil(SM->computeReciprocalThroughput(Prev->getInstr())));
SDep D(Prev, SDep::Artificial);
- D.setLatency(*LDSBandwidth);
+ D.setLatency(Spacing);
DAG->addEdge(LI.SU, D);
LLVM_DEBUG(dbgs() << "[3+4] ds_load SU" << Prev->NodeNum << " -> SU"
- << LI.SU->NodeNum << " (spacing latency "
- << (unsigned)*LDSBandwidth << ")\n");
+ << LI.SU->NodeNum << " (spacing latency " << Spacing
+ << ")\n");
}
Prev = LI.SU;
}
@@ -236,22 +231,26 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
"honor the ds_load -> ds_load spacing.\n");
for (LoadInfo &LI : Loads)
if (LI.MinPos != UINT_MAX) {
- LI.LatestCycle = (long)LI.MinPos * (*WmmaLatency) - (long)(*LoadLatency);
+ const long LoadLatency =
+ static_cast<long>(SM->computeInstrLatency(LI.SU->getInstr()));
+ LI.LatestCycle = (long)LI.MinPos * (*WmmaLatency) - LoadLatency;
LLVM_DEBUG(dbgs() << "[lat] ds_load SU" << LI.SU->NodeNum
<< ": LatestCycle=" << LI.LatestCycle << " (W["
<< LI.MinPos << "]*" << *WmmaLatency << " - "
- << *LoadLatency << ")\n");
+ << LoadLatency << ")\n");
}
long PrevLatest = LONG_MAX;
for (int I = (int)Loads.size() - 1; I >= 0; --I) {
LoadInfo &LI = Loads[I];
if (LI.MinPos == UINT_MAX)
continue;
- long Spaced = PrevLatest - (long)(*LDSBandwidth);
+ const long Spacing = static_cast<long>(
+ std::ceil(SM->computeReciprocalThroughput(LI.SU->getInstr())));
+ long Spaced = PrevLatest - Spacing;
if (Spaced < LI.LatestCycle) {
LLVM_DEBUG(dbgs() << "[space] ds_load SU" << LI.SU->NodeNum
<< ": LatestCycle " << LI.LatestCycle << " -> "
- << Spaced << " (spaced " << (unsigned)*LDSBandwidth
+ << Spaced << " (spaced " << Spacing
<< " before next load's " << PrevLatest << ")\n");
LI.LatestCycle = Spaced;
}
>From 66416d793427fbc1d91464cbfa64b394f21c4bda Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Wed, 9 Sep 2026 18:49:27 -0600
Subject: [PATCH 17/20] redundant check
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 84acbc3624ab2..dcb5ed178e77d 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -106,7 +106,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
}
// The following means the DAG Mutation cannot do anything useful.
- if (Loads.empty() || !WmmaLatency || Wmmas.empty())
+ if (Loads.empty() || Wmmas.empty())
return;
LLVM_DEBUG(
>From 9f96ba18df29a9ba237ee0dc2c9e23bb85dd5163 Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Wed, 9 Sep 2026 19:02:12 -0600
Subject: [PATCH 18/20] Covered various nits - Replaced C-style casts with
`static_cast` - Replaced reverse loop with `llvm::reverse` - Used
destructuring for loops over `Frags` - Replaced loop accumulating max element
with `llvm::max_element` - Removed unused CommandLine.h inclusion.
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 35 ++++++++-----------
1 file changed, 14 insertions(+), 21 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index dcb5ed178e77d..81aa9a1bf4fe4 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -31,7 +31,6 @@
#include "SIInstrInfo.h"
#include "llvm/CodeGen/ScheduleDAG.h"
#include "llvm/CodeGen/ScheduleDAGInstrs.h"
-#include "llvm/Support/CommandLine.h"
#include "llvm/Support/Debug.h"
#include <optional>
#define DEBUG_TYPE "amdgpu-wmma-sched"
@@ -233,15 +232,15 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (LI.MinPos != UINT_MAX) {
const long LoadLatency =
static_cast<long>(SM->computeInstrLatency(LI.SU->getInstr()));
- LI.LatestCycle = (long)LI.MinPos * (*WmmaLatency) - LoadLatency;
+ LI.LatestCycle =
+ static_cast<long>(LI.MinPos) * (*WmmaLatency) - LoadLatency;
LLVM_DEBUG(dbgs() << "[lat] ds_load SU" << LI.SU->NodeNum
<< ": LatestCycle=" << LI.LatestCycle << " (W["
<< LI.MinPos << "]*" << *WmmaLatency << " - "
<< LoadLatency << ")\n");
}
long PrevLatest = LONG_MAX;
- for (int I = (int)Loads.size() - 1; I >= 0; --I) {
- LoadInfo &LI = Loads[I];
+ for (LoadInfo &LI : llvm::reverse(Loads)) {
if (LI.MinPos == UINT_MAX)
continue;
const long Spacing = static_cast<long>(
@@ -283,10 +282,9 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
}
std::vector<unsigned> Hist(Wmmas.size(), 0);
- for (auto &KV : Frags) {
- FragInfo &F = KV.second;
- long Pos = F.LatestCycle / (long)(*WmmaLatency);
- unsigned StartPos = Pos < 0 ? 0 : (unsigned)Pos;
+ for (auto &[_, F] : Frags) {
+ long Pos = F.LatestCycle / static_cast<long>(*WmmaLatency);
+ unsigned StartPos = Pos < 0 ? 0 : static_cast<unsigned>(Pos);
for (unsigned P = StartPos; P <= F.MaxPos && P < Wmmas.size(); ++P)
Hist[P] += F.VGPRs;
LLVM_DEBUG({
@@ -298,9 +296,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
});
}
- unsigned Budget = 0;
- for (unsigned P = 0; P < Wmmas.size(); ++P)
- Budget = std::max(Budget, Hist[P]);
+ unsigned Budget = *llvm::max_element(Hist);
LLVM_DEBUG({
dbgs() << "\n--- [5] histogram BEFORE debunch (sets the budget) -------\n"
@@ -328,14 +324,13 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
"any earlier. 'unconstrained' = it already fits at W[0], so no edge\n"
"is needed; the histogram is updated cumulatively so later\n"
"fragments only use the slack that's left over.\n");
- for (auto &KV : Frags) {
- FragInfo &F = KV.second;
- long Pos = F.LatestCycle / (long)(*WmmaLatency);
- unsigned LateStartPos = Pos < 0 ? 0 : (unsigned)Pos;
+ for (auto &[_, F] : Frags) {
+ long Pos = F.LatestCycle / static_cast<long>(*WmmaLatency);
+ unsigned LateStartPos = Pos < 0 ? 0 : static_cast<unsigned>(Pos);
unsigned Earliest = LateStartPos;
- for (int P = (int)LateStartPos - 1; P >= 0; --P) {
- if (Hist[(unsigned)P] + F.VGPRs <= Budget)
- Earliest = (unsigned)P;
+ for (int P = static_cast<int>(LateStartPos) - 1; P >= 0; --P) {
+ if (Hist[static_cast<unsigned>(P)] + F.VGPRs <= Budget)
+ Earliest = static_cast<unsigned>(P);
else
break;
}
@@ -359,9 +354,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
}
LLVM_DEBUG({
- unsigned Peak = 0;
- for (unsigned P = 0; P < Wmmas.size(); ++P)
- Peak = std::max(Peak, Hist[P]);
+ unsigned Peak = *llvm::max_element(Hist);
dbgs() << "\n--- [6] histogram AFTER debunch -------------------------\n"
"Same usage after the debunch edges. Loads now sit as\n"
"early as the budget allows; the peak should still be within\n"
>From 1aa4f7d28becdee7a24e4835ba37e35a01e4c993 Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Wed, 9 Sep 2026 19:27:43 -0600
Subject: [PATCH 19/20] Replaced usage of `MapVector` with `SmallVector`
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 23 ++++++++++---------
1 file changed, 12 insertions(+), 11 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 81aa9a1bf4fe4..3f3ed5b512aa6 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -82,7 +82,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
const SIInstrInfo *TII = ST.getInstrInfo();
// Gather WMMAs (numbered in program order) and ds_loads.
- MapVector<SUnit *, unsigned> Wmmas; // Ordered WMMA SUnits
+ SmallVector<SUnit *> Wmmas; // Ordered WMMA SUnits
SmallVector<LoadInfo> Loads;
std::optional<unsigned> WmmaLatency;
@@ -95,7 +95,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (TII->isMFMAorWMMA(*MI)) {
if (!WmmaLatency)
WmmaLatency = SM->computeInstrLatency(MI);
- Wmmas.insert({&SU, Wmmas.size()});
+ Wmmas.push_back(&SU);
continue;
}
@@ -128,14 +128,14 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
"Chain each WMMA to the next in program order with an artificial\n"
"edge, so load placement can be reasoned about relative to fixed\n"
"W[] positions.\n");
- auto [PrevSU, PrevPos] = *Wmmas.begin();
- for (auto *It = std::next(Wmmas.begin()); It != Wmmas.end(); ++It) {
- auto [SU, Pos] = *It;
+ SUnit *PrevSU = Wmmas.front();
+ for (unsigned Pos = 1; Pos < Wmmas.size(); ++Pos) {
+ SUnit *SU = Wmmas[Pos];
+ unsigned PrevPos = Pos - 1;
DAG->addEdge(SU, SDep(PrevSU, SDep::Artificial));
LLVM_DEBUG(dbgs() << "[1] WMMA W[" << PrevPos << "] SU" << PrevSU->NodeNum
<< " -> W[" << Pos << "] SU" << SU->NodeNum << "\n");
PrevSU = SU;
- PrevPos = Pos;
}
// For each load, find earliest and latest consuming WMMA positions, and
@@ -153,17 +153,18 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
for (const SDep &D : LI.SU->Succs) {
if (D.getKind() != SDep::Data)
continue;
- auto *It = Wmmas.find(D.getSUnit());
+ auto It = llvm::find(Wmmas, D.getSUnit());
// Check to see if successor is a WMMA
if (It == Wmmas.end())
continue;
- LI.MinPos = std::min(LI.MinPos, It->second);
- LI.MaxPos = std::max(LI.MaxPos, It->second);
+ unsigned Pos = static_cast<unsigned>(std::distance(Wmmas.begin(), It));
+ LI.MinPos = std::min(LI.MinPos, Pos);
+ LI.MaxPos = std::max(LI.MaxPos, Pos);
Consumers.push_back(D.getSUnit());
}
if (LI.MinPos == UINT_MAX)
continue;
- SUnit *EarliestConsumer = Wmmas.begin()[LI.MinPos].first;
+ SUnit *EarliestConsumer = Wmmas[LI.MinPos];
const unsigned LoadLatency = SM->computeInstrLatency(LI.SU->getInstr());
// Correct latency of edges between ds_load and earliest WMMA consumer
for (SDep &S : LI.SU->Succs)
@@ -350,7 +351,7 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
if (Earliest == 0)
continue;
for (SUnit *L : F.Subloads)
- DAG->addEdge(L, SDep(Wmmas.begin()[Earliest].first, SDep::Artificial));
+ DAG->addEdge(L, SDep(Wmmas[Earliest], SDep::Artificial));
}
LLVM_DEBUG({
>From 7689c600c9359d3fde0bf45508747ed8c99731c4 Mon Sep 17 00:00:00 2001
From: Axel Sorenson <AxelPSorenson at gmail.com>
Date: Wed, 9 Sep 2026 19:42:32 -0600
Subject: [PATCH 20/20] Now filters out loads without a `MinPos` with
`erase_if` and no longer finds `LatestCycle` in a separate loop.
---
llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp | 41 +++++++------------
1 file changed, 15 insertions(+), 26 deletions(-)
diff --git a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
index 3f3ed5b512aa6..3355be8d5efd3 100644
--- a/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
+++ b/llvm/lib/Target/AMDGPU/AMDGPUWMMASchedule.cpp
@@ -147,7 +147,8 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
"For each ds_load, find which WMMAs consume it (MinPos = earliest,\n"
"MaxPos = latest). Update the data edge latency of the earliest\n"
"consumer to be the real LDS load latency, so the scheduler keeps\n"
- "the load issued far enough ahead of the WMMAs that need it.\n");
+ "the load issued far enough ahead of the WMMAs that need it. Also\n"
+ "calculate the latest cycle at which the load can be issued.\n");
for (LoadInfo &LI : Loads) {
SmallVector<SUnit *, 8> Consumers; // WMMA consumers (program order)
for (const SDep &D : LI.SU->Succs) {
@@ -175,6 +176,8 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
P.setLatency(LoadLatency);
EarliestConsumer->setDepthDirty();
LI.SU->setHeightDirty();
+ LI.LatestCycle = static_cast<long>(LI.MinPos) * (*WmmaLatency) -
+ static_cast<long>(LoadLatency);
LLVM_DEBUG({
dbgs() << "[2] ds_load SU" << LI.SU->NodeNum << ": MinPos=" << LI.MinPos
<< " MaxPos=" << LI.MaxPos << "; consumers W[" << LI.MinPos << ".."
@@ -183,9 +186,16 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
dbgs() << (I ? ", " : "") << "SU" << Consumers[I]->NodeNum;
dbgs() << "); set latency " << LoadLatency << " on edge -> W["
<< LI.MinPos << "]\n";
+ dbgs() << "[lat] ds_load SU" << LI.SU->NodeNum
+ << ": LatestCycle=" << LI.LatestCycle << " (W[" << LI.MinPos
+ << "]*" << *WmmaLatency << " - " << LoadLatency << ")\n";
});
}
+ // Loads without a MinPos have no WMMA consumer in this scheduling region.
+ llvm::erase_if(Loads,
+ [](const LoadInfo &LI) { return LI.MinPos == UINT_MAX; });
+
// Order the Loads
llvm::stable_sort(Loads, [](const LoadInfo &A, const LoadInfo &B) {
return A.MinPos < B.MinPos;
@@ -201,8 +211,6 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
"memory bound).\n");
SUnit *Prev = nullptr;
for (LoadInfo &LI : Loads) {
- if (LI.MinPos == UINT_MAX)
- continue;
if (Prev) {
const unsigned Spacing = static_cast<unsigned>(
std::ceil(SM->computeReciprocalThroughput(Prev->getInstr())));
@@ -216,34 +224,17 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
Prev = LI.SU;
}
- // Determing each load's as late as possible cycle - this means the
- // latest cycle that still meets the load latency, then pushed earlier
- // if the ds_load -> ds_load edges requires spacing (ds_loads cannot be
- // too close to each other or it could overwhelm the LDS bus and lead to
- // the program being memory bound).
+ // Push each load's latest cycle earlier if the ds_load -> ds_load edges
+ // require spacing (ds_loads cannot be too close to each other or they could
+ // overwhelm the LDS bus and make the program memory bound).
LLVM_DEBUG(
dbgs()
- << "\n--- [lat]/[space] as late as possible cycle --------------\n"
- "[lat]: latest cycle each load could issue and still feed its\n"
- "earliest consumer in time (MinPos*wmmalat - loadlat).\n"
+ << "\n--- [space] as late as possible cycle --------------------\n"
"[space]: loop through the ordered loads from last to first and\n"
"pull any that are too close to the next one earlier in order to\n"
"honor the ds_load -> ds_load spacing.\n");
- for (LoadInfo &LI : Loads)
- if (LI.MinPos != UINT_MAX) {
- const long LoadLatency =
- static_cast<long>(SM->computeInstrLatency(LI.SU->getInstr()));
- LI.LatestCycle =
- static_cast<long>(LI.MinPos) * (*WmmaLatency) - LoadLatency;
- LLVM_DEBUG(dbgs() << "[lat] ds_load SU" << LI.SU->NodeNum
- << ": LatestCycle=" << LI.LatestCycle << " (W["
- << LI.MinPos << "]*" << *WmmaLatency << " - "
- << LoadLatency << ")\n");
- }
long PrevLatest = LONG_MAX;
for (LoadInfo &LI : llvm::reverse(Loads)) {
- if (LI.MinPos == UINT_MAX)
- continue;
const long Spacing = static_cast<long>(
std::ceil(SM->computeReciprocalThroughput(LI.SU->getInstr())));
long Spaced = PrevLatest - Spacing;
@@ -270,8 +261,6 @@ void WMMASchedule::apply(ScheduleDAGInstrs *DAG) {
"as late as possible schedule above.\n");
MapVector<Register, FragInfo> Frags;
for (LoadInfo &LI : Loads) {
- if (LI.MinPos == UINT_MAX)
- continue;
Register R = LI.SU->getInstr()->getOperand(0).getReg();
FragInfo &F = Frags[R];
if (F.Subloads.empty() && R.isVirtual())
More information about the llvm-commits
mailing list