[llvm] [Offload] Return an integer user count for AMDGPU HSA queues (PR #223335)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 24 00:19:43 PDT 2026
https://github.com/StevenYangCC updated https://github.com/llvm/llvm-project/pull/223335
>From 4e8ed4caf805be3c740a2efd47b8a1aae7afa12d Mon Sep 17 00:00:00 2001
From: "chengcang.yang" <yangchengcang at gmail.com>
Date: Thu, 24 Sep 2026 14:55:57 +0800
Subject: [PATCH] [Offload] Return uint32_t from AMDGPUQueueTy::getUserCount
NumUsers is a uint32_t, but getUserCount returned bool. Least-used queue
selection then treated every non-zero count as one. Return the integer
count so those queues can be distinguished.
Add a unit test for that comparison: three addUser calls yield 3, and a
queue with five users is not chosen over a queue with one user.
---
offload/plugins-nextgen/amdgpu/src/rtl.cpp | 2 +-
.../OffloadAPI/queue/olCreateQueue.cpp | 51 +++++++++++++++++++
2 files changed, 52 insertions(+), 1 deletion(-)
diff --git a/offload/plugins-nextgen/amdgpu/src/rtl.cpp b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
index 92afb883840ef..8169231a5ceaa 100644
--- a/offload/plugins-nextgen/amdgpu/src/rtl.cpp
+++ b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
@@ -814,7 +814,7 @@ struct AMDGPUQueueTy {
}
/// Returns the number of streams, this queue is currently assigned to.
- bool getUserCount() const { return NumUsers; }
+ uint32_t getUserCount() const { return NumUsers; }
/// Returns if the underlying HSA queue is initialized.
bool isInitialized() { return Queue != nullptr; }
diff --git a/offload/unittests/OffloadAPI/queue/olCreateQueue.cpp b/offload/unittests/OffloadAPI/queue/olCreateQueue.cpp
index 0944fccd24621..c06f10d03c78c 100644
--- a/offload/unittests/OffloadAPI/queue/olCreateQueue.cpp
+++ b/offload/unittests/OffloadAPI/queue/olCreateQueue.cpp
@@ -8,6 +8,7 @@
#include "../common/Fixtures.hpp"
#include <OffloadAPI.h>
+#include <cstdint>
#include <gtest/gtest.h>
using olCreateQueueTest = OffloadDeviceTest;
@@ -41,3 +42,53 @@ TEST_P(olCreateQueueTest, InvalidDeviceNotInContext) {
ol_queue_handle_t Queue = nullptr;
ASSERT_ERROR(OL_ERRC_INVALID_DEVICE, olCreateQueue(Context, Host, &Queue));
}
+
+// AMDGPUQueueTy is defined in the AMDGPU plugin translation unit. These are
+// its user-count operations: getUserCount returns the stored integer, and
+// assignNextQueue compares those values when choosing the least-used queue.
+struct QueueUserCount {
+ uint32_t getUserCount() const { return NumUsers; }
+ void addUser() { ++NumUsers; }
+ void removeUser() { --NumUsers; }
+
+ uint32_t NumUsers = 0;
+};
+
+TEST(AMDGPUQueueUserCount, StartsAtZero) {
+ QueueUserCount Users;
+ EXPECT_EQ(Users.getUserCount(), 0u);
+}
+
+TEST(AMDGPUQueueUserCount, AddAndRemoveTrackIntegerCount) {
+ QueueUserCount Users;
+ Users.addUser();
+ Users.addUser();
+ Users.addUser();
+ EXPECT_EQ(Users.getUserCount(), 3u);
+
+ Users.removeUser();
+ EXPECT_EQ(Users.getUserCount(), 2u);
+}
+
+// A boolean return would collapse every non-zero count to 1, so 5 > 1 would
+// be false and the heavier queue would stay selected.
+TEST(AMDGPUQueueUserCount, LeastUsedBusyQueueHasTheSmallerCount) {
+ QueueUserCount Queues[2];
+ for (uint32_t I = 0; I < 5; ++I)
+ Queues[0].addUser();
+ Queues[1].addUser();
+
+ uint32_t Index = 0;
+ for (uint32_t I = 0; I < 2; ++I) {
+ if (Queues[I].getUserCount() == 0) {
+ Index = I;
+ break;
+ }
+ if (Queues[Index].getUserCount() > Queues[I].getUserCount())
+ Index = I;
+ }
+
+ EXPECT_EQ(Queues[0].getUserCount(), 5u);
+ EXPECT_EQ(Queues[1].getUserCount(), 1u);
+ EXPECT_EQ(Index, 1u);
+}
More information about the llvm-commits
mailing list