[llvm] [SelectionDAG] Reduce RRList scheduling overhead (PR #223241)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 28 21:57:13 PDT 2026
https://github.com/1sgtpepper updated https://github.com/llvm/llvm-project/pull/223241
>From 9ace50d19f0306c20ef2dbfb051b4035e446af5c Mon Sep 17 00:00:00 2001
From: 1sgtpepper <cynejarviszarceno at gmail.com>
Date: Sun, 13 Sep 2026 21:02:54 +0800
Subject: [PATCH 1/3] [SelectionDAG] Reduce RRList scheduling overhead
Cache closest-successor distances while nodes remain in the available
queue, invalidating them at the queue lifecycle boundaries.
Replace the assertion-only Sethi-Ullman worklist membership scan with
a depth bound to avoid quadratic work on long dependency paths.
Fixes #211018.
---
.../SelectionDAG/ScheduleDAGRRList.cpp | 30 ++++++++++++++-----
1 file changed, 23 insertions(+), 7 deletions(-)
diff --git a/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp b/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp
index 520fe43a8cfc2..c08a25ff1ff20 100644
--- a/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp
@@ -1716,6 +1716,10 @@ class RegReductionPQBase : public SchedulingPriorityQueue {
// SethiUllmanNumbers - The SethiUllman number for each node.
std::vector<unsigned> SethiUllmanNumbers;
+ // A queued node's successors are already scheduled. Their heights remain
+ // fixed until the node leaves the queue or its dependencies are updated.
+ DenseMap<const SUnit *, unsigned> ClosestSuccs;
+
/// RegPressure - Tracking current reg pressure per register class.
std::vector<unsigned> RegPressure;
@@ -1761,11 +1765,14 @@ class RegReductionPQBase : public SchedulingPriorityQueue {
void releaseState() override {
SUnits = nullptr;
SethiUllmanNumbers.clear();
+ ClosestSuccs.clear();
llvm::fill(RegPressure, 0);
}
unsigned getNodePriority(const SUnit *SU) const;
+ unsigned getClosestSucc(const SUnit *SU);
+
unsigned getNodeOrdering(const SUnit *SU) const {
if (!SU->getNode()) return 0;
@@ -1787,6 +1794,7 @@ class RegReductionPQBase : public SchedulingPriorityQueue {
if (I != std::prev(Queue.end()))
std::swap(*I, Queue.back());
Queue.pop_back();
+ ClosestSuccs.erase(SU);
SU->NodeQueueId = 0;
}
@@ -1871,6 +1879,7 @@ class RegReductionPriorityQueue : public RegReductionPQBase {
if (Queue.empty()) return nullptr;
SUnit *V = popFromQueue(Queue, Picker, scheduleDAG);
+ ClosestSuccs.erase(V);
V->NodeQueueId = 0;
return V;
}
@@ -1940,11 +1949,10 @@ CalcNodeSethiUllmanNumber(const SUnit *SU, std::vector<unsigned> &SUNumbers) {
if (Pred.isCtrl()) continue; // ignore chain preds
SUnit *PredSU = Pred.getSUnit();
if (SUNumbers[PredSU->NodeNum] == 0) {
-#ifndef NDEBUG
- // In debug mode, check that we don't have such element in the stack.
- for (auto It : WorkList)
- assert(It.SU != PredSU && "Trying to push an element twice?");
-#endif
+ // An acyclic path cannot contain more nodes than SUNumbers. Avoid
+ // scanning the worklist, which is quadratic for long paths.
+ assert(WorkList.size() < SUNumbers.size() &&
+ "Trying to push an element twice?");
// Next time start processing this one starting from the next pred.
Temp.PredsProcessed = P + 1;
WorkList.push_back(PredSU);
@@ -1999,6 +2007,7 @@ void RegReductionPQBase::addNode(const SUnit *SU) {
}
void RegReductionPQBase::updateNode(const SUnit *SU) {
+ ClosestSuccs.erase(SU);
SethiUllmanNumbers[SU->NodeNum] = 0;
CalcNodeSethiUllmanNumber(SU, SethiUllmanNumbers);
}
@@ -2327,6 +2336,13 @@ static unsigned closestSucc(const SUnit *SU) {
return MaxHeight;
}
+unsigned RegReductionPQBase::getClosestSucc(const SUnit *SU) {
+ auto It = ClosestSuccs.find(SU);
+ if (It != ClosestSuccs.end())
+ return It->second;
+ return ClosestSuccs.try_emplace(SU, closestSucc(SU)).first->second;
+}
+
/// calcMaxScratches - Returns an cost estimate of the worse case requirement
/// for scratch registers, i.e. number of data dependencies.
static unsigned calcMaxScratches(const SUnit *SU) {
@@ -2573,8 +2589,8 @@ static bool BURRSort(SUnit *left, SUnit *right, RegReductionPQBase *SPQ) {
// t3 = op t4, c2
//
// This creates more short live intervals.
- unsigned LDist = closestSucc(left);
- unsigned RDist = closestSucc(right);
+ unsigned LDist = SPQ->getClosestSucc(left);
+ unsigned RDist = SPQ->getClosestSucc(right);
if (LDist != RDist)
return LDist < RDist;
>From 46fa7c3f4fd176b59f6c2077724b6bcbfd05a338 Mon Sep 17 00:00:00 2001
From: 1sgtpepper <cynejarviszarceno at gmail.com>
Date: Mon, 14 Sep 2026 17:34:31 +0800
Subject: [PATCH 2/3] [SelectionDAG] Address RRList scheduling review feedback
---
.../SelectionDAG/ScheduleDAGRRList.cpp | 8 ++++----
llvm/test/CodeGen/NVPTX/bug211018.ll | 20 +++++++++++++++++++
2 files changed, 24 insertions(+), 4 deletions(-)
create mode 100644 llvm/test/CodeGen/NVPTX/bug211018.ll
diff --git a/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp b/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp
index c08a25ff1ff20..4b8158eadc4b5 100644
--- a/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp
+++ b/llvm/lib/CodeGen/SelectionDAG/ScheduleDAGRRList.cpp
@@ -2337,10 +2337,10 @@ static unsigned closestSucc(const SUnit *SU) {
}
unsigned RegReductionPQBase::getClosestSucc(const SUnit *SU) {
- auto It = ClosestSuccs.find(SU);
- if (It != ClosestSuccs.end())
- return It->second;
- return ClosestSuccs.try_emplace(SU, closestSucc(SU)).first->second;
+ auto [It, Inserted] = ClosestSuccs.try_emplace(SU);
+ if (Inserted)
+ It->second = closestSucc(SU);
+ return It->second;
}
/// calcMaxScratches - Returns an cost estimate of the worse case requirement
diff --git a/llvm/test/CodeGen/NVPTX/bug211018.ll b/llvm/test/CodeGen/NVPTX/bug211018.ll
new file mode 100644
index 0000000000000..9d5ea3eca313f
--- /dev/null
+++ b/llvm/test/CodeGen/NVPTX/bug211018.ll
@@ -0,0 +1,20 @@
+; RUN: llc < %s -O2 -filetype=null -o -
+
+; Keep the issue reproducer as compile-only coverage. Compile-time comparisons
+; are performed separately because a timing threshold would be host-dependent.
+; https://github.com/llvm/llvm-project/issues/211018
+
+target datalayout = "e-p:64:64-i64:64-i128:128-v16:16-v32:32-n16:32:64-S32"
+target triple = "nvptx64-nvidia-cuda"
+
+declare float @llvm.vector.reduce.fadd.v65536f32(float, <65536 x float>)
+
+define void @kernel(ptr addrspace(1) %out, ptr addrspace(1) %in) {
+entry:
+ %val = load <65536 x float>, ptr addrspace(1) %in, align 32
+ %mul = fmul <65536 x float> %val, %val
+ %res = call float @llvm.vector.reduce.fadd.v65536f32(
+ float 0.000000e+00, <65536 x float> %mul)
+ store float %res, ptr addrspace(1) %out, align 4
+ ret void
+}
>From bac562fae422f2a2f1ec55b7ba7731da5cb4d967 Mon Sep 17 00:00:00 2001
From: 1sgtpepper <cynejarviszarceno at gmail.com>
Date: Mon, 14 Sep 2026 21:30:00 +0800
Subject: [PATCH 3/3] [NVPTX] Simplify bug211018 test
---
llvm/test/CodeGen/NVPTX/bug211018.ll | 3 +--
1 file changed, 1 insertion(+), 2 deletions(-)
diff --git a/llvm/test/CodeGen/NVPTX/bug211018.ll b/llvm/test/CodeGen/NVPTX/bug211018.ll
index 9d5ea3eca313f..4024dc7179507 100644
--- a/llvm/test/CodeGen/NVPTX/bug211018.ll
+++ b/llvm/test/CodeGen/NVPTX/bug211018.ll
@@ -1,10 +1,9 @@
-; RUN: llc < %s -O2 -filetype=null -o -
+; RUN: llc < %s -filetype=null -o -
; Keep the issue reproducer as compile-only coverage. Compile-time comparisons
; are performed separately because a timing threshold would be host-dependent.
; https://github.com/llvm/llvm-project/issues/211018
-target datalayout = "e-p:64:64-i64:64-i128:128-v16:16-v32:32-n16:32:64-S32"
target triple = "nvptx64-nvidia-cuda"
declare float @llvm.vector.reduce.fadd.v65536f32(float, <65536 x float>)
More information about the llvm-commits
mailing list