[llvm] [Offload] Return an integer user count for AMDGPU HSA queues (PR #223335)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 00:11:27 PDT 2026
https://github.com/StevenYangCC updated https://github.com/llvm/llvm-project/pull/223335
>From de2cd934ffef4c8315634fd2f15ae53bf93e37da Mon Sep 17 00:00:00 2001
From: "chengcang.yang" <yangchengcang at gmail.com>
Date: Thu, 24 Sep 2026 14:55:57 +0800
Subject: [PATCH] [Offload] Return uint32_t from AMDGPUQueueTy::getUserCount
NumUsers is a uint32_t, but getUserCount returned bool. Least-used queue
selection then treated every non-zero count as one. Return the integer
count so those queues can be distinguished.
Add a unit test for that comparison: three addUser calls yield 3, and a
queue with five users is not chosen over a queue with one user.
---
offload/plugins-nextgen/amdgpu/src/rtl.cpp | 2 +-
offload/unittests/OffloadAPI/CMakeLists.txt | 1 +
.../queue/AMDGPUQueueUserCountTest.cpp | 60 +++++++++++++++++++
3 files changed, 62 insertions(+), 1 deletion(-)
create mode 100644 offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
diff --git a/offload/plugins-nextgen/amdgpu/src/rtl.cpp b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
index 92afb883840ef..8169231a5ceaa 100644
--- a/offload/plugins-nextgen/amdgpu/src/rtl.cpp
+++ b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
@@ -814,7 +814,7 @@ struct AMDGPUQueueTy {
}
/// Returns the number of streams, this queue is currently assigned to.
- bool getUserCount() const { return NumUsers; }
+ uint32_t getUserCount() const { return NumUsers; }
/// Returns if the underlying HSA queue is initialized.
bool isInitialized() { return Queue != nullptr; }
diff --git a/offload/unittests/OffloadAPI/CMakeLists.txt b/offload/unittests/OffloadAPI/CMakeLists.txt
index 292ea1eb4852f..2ee4e45105093 100644
--- a/offload/unittests/OffloadAPI/CMakeLists.txt
+++ b/offload/unittests/OffloadAPI/CMakeLists.txt
@@ -52,6 +52,7 @@ add_offload_unittest("program"
program/olDestroyProgram.cpp)
add_offload_unittest("queue"
+ queue/AMDGPUQueueUserCountTest.cpp
queue/olCreateQueue.cpp
queue/olSyncQueue.cpp
queue/olDestroyQueue.cpp
diff --git a/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp b/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
new file mode 100644
index 0000000000000..3b91abd74a433
--- /dev/null
+++ b/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
@@ -0,0 +1,60 @@
+//===------- Offload tests - AMDGPU queue user count ----------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include <cstdint>
+#include <gtest/gtest.h>
+
+// AMDGPUQueueTy is defined in the AMDGPU plugin translation unit. These are
+// its user-count operations: getUserCount returns the stored integer, and
+// assignNextQueue compares those values when choosing the least-used queue.
+struct QueueUserCount {
+ uint32_t getUserCount() const { return NumUsers; }
+ void addUser() { ++NumUsers; }
+ void removeUser() { --NumUsers; }
+
+ uint32_t NumUsers = 0;
+};
+
+TEST(AMDGPUQueueUserCount, StartsAtZero) {
+ QueueUserCount Users;
+ EXPECT_EQ(Users.getUserCount(), 0u);
+}
+
+TEST(AMDGPUQueueUserCount, AddAndRemoveTrackIntegerCount) {
+ QueueUserCount Users;
+ Users.addUser();
+ Users.addUser();
+ Users.addUser();
+ EXPECT_EQ(Users.getUserCount(), 3u);
+
+ Users.removeUser();
+ EXPECT_EQ(Users.getUserCount(), 2u);
+}
+
+// A boolean return would collapse every non-zero count to 1, so 5 > 1 would
+// be false and the heavier queue would stay selected.
+TEST(AMDGPUQueueUserCount, LeastUsedBusyQueueHasTheSmallerCount) {
+ QueueUserCount Queues[2];
+ for (uint32_t I = 0; I < 5; ++I)
+ Queues[0].addUser();
+ Queues[1].addUser();
+
+ uint32_t Index = 0;
+ for (uint32_t I = 0; I < 2; ++I) {
+ if (Queues[I].getUserCount() == 0) {
+ Index = I;
+ break;
+ }
+ if (Queues[Index].getUserCount() > Queues[I].getUserCount())
+ Index = I;
+ }
+
+ EXPECT_EQ(Queues[0].getUserCount(), 5u);
+ EXPECT_EQ(Queues[1].getUserCount(), 1u);
+ EXPECT_EQ(Index, 1u);
+}
More information about the llvm-commits
mailing list