[llvm] [Offload] Return an integer user count for AMDGPU HSA queues (PR #223335)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Sep 17 04:07:12 PDT 2026
https://github.com/StevenYangCC updated https://github.com/llvm/llvm-project/pull/223335
>From 8eeb2a2d76bc55973dc437b6f1743dfb86c6e61e Mon Sep 17 00:00:00 2001
From: "chengcang.yang" <yangchengcang at gmail.com>
Date: Mon, 14 Sep 2026 17:08:33 +0800
Subject: [PATCH] [Offload] Return an integer user count for AMDGPU HSA queues
getUserCount returned bool while the stored count is uint32_t, so
busy-queue selection treated every non-zero count as one. Return the
integer count so least-used assignment can distinguish differently
loaded queues.
---
offload/plugins-nextgen/amdgpu/src/rtl.cpp | 15 ++++---
.../amdgpu/utils/AMDGPUQueueUserCount.h | 34 ++++++++++++++
offload/unittests/OffloadAPI/CMakeLists.txt | 3 ++
.../queue/AMDGPUQueueUserCountTest.cpp | 44 +++++++++++++++++++
4 files changed, 89 insertions(+), 7 deletions(-)
create mode 100644 offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h
create mode 100644 offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
diff --git a/offload/plugins-nextgen/amdgpu/src/rtl.cpp b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
index 281b9e3795a54..9cbc52e866829 100644
--- a/offload/plugins-nextgen/amdgpu/src/rtl.cpp
+++ b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
@@ -30,6 +30,7 @@
#include "Shared/Utils.h"
#include "Utils/ELF.h"
+#include "AMDGPUQueueUserCount.h"
#include "GlobalHandler.h"
#include "OffloadAPI.h"
#include "OpenMP/OMPT/Callback.h"
@@ -780,7 +781,7 @@ using AMDGPUSignalManagerTy = GenericDeviceResourceManagerTy<AMDGPUSignalRef>;
/// Class holding an HSA queue to submit kernel and barrier packets.
struct AMDGPUQueueTy {
/// Create an empty queue.
- AMDGPUQueueTy() : Queue(nullptr), Mutex(), NumUsers(0) {}
+ AMDGPUQueueTy() : Queue(nullptr), Mutex() {}
/// Lazily initialize a new queue belonging to a specific agent.
Error init(GenericDeviceTy &Device, hsa_agent_t Agent, int32_t QueueSize) {
@@ -813,17 +814,17 @@ struct AMDGPUQueueTy {
return Plugin::check(Status, "error in hsa_queue_destroy: %s");
}
- /// Returns the number of streams, this queue is currently assigned to.
- bool getUserCount() const { return NumUsers; }
+ /// Returns the number of streams this queue is currently assigned to.
+ uint32_t getUserCount() const { return UserCount.getUserCount(); }
/// Returns if the underlying HSA queue is initialized.
bool isInitialized() { return Queue != nullptr; }
/// Decrement user count of the queue object.
- void removeUser() { --NumUsers; }
+ void removeUser() { UserCount.removeUser(); }
/// Increase user count of the queue object.
- void addUser() { ++NumUsers; }
+ void addUser() { UserCount.addUser(); }
/// Push a kernel launch to the queue. The kernel launch requires an output
/// signal and can define an optional input signal (nullptr if none).
@@ -1006,9 +1007,9 @@ struct AMDGPUQueueTy {
/// atomic operations. We can further investigate it if this is a bottleneck.
std::mutex Mutex;
- /// The number of streams, this queue is currently assigned to. A queue is
+ /// Tracks how many streams this queue is currently assigned to. A queue is
/// considered idle when this is zero, otherwise: busy.
- uint32_t NumUsers;
+ AMDGPUQueueUserCount UserCount;
};
/// Struct that implements a stream of asynchronous operations for AMDGPU
diff --git a/offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h b/offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h
new file mode 100644
index 0000000000000..67428ce60155a
--- /dev/null
+++ b/offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h
@@ -0,0 +1,34 @@
+//===- AMDGPUQueueUserCount.h - HSA queue stream user count -----*- C++ -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef OFFLOAD_PLUGINS_NEXTGEN_AMDGPU_UTILS_AMDGPUQUEUEUSERCOUNT_H
+#define OFFLOAD_PLUGINS_NEXTGEN_AMDGPU_UTILS_AMDGPUQUEUEUSERCOUNT_H
+
+#include <cstdint>
+
+namespace llvm {
+namespace omp {
+namespace target {
+namespace plugin {
+
+/// Tracks how many streams currently use an HSA queue.
+struct AMDGPUQueueUserCount {
+ uint32_t getUserCount() const { return NumUsers; }
+ void addUser() { ++NumUsers; }
+ void removeUser() { --NumUsers; }
+
+private:
+ uint32_t NumUsers = 0;
+};
+
+} // namespace plugin
+} // namespace target
+} // namespace omp
+} // namespace llvm
+
+#endif // OFFLOAD_PLUGINS_NEXTGEN_AMDGPU_UTILS_AMDGPUQUEUEUSERCOUNT_H
diff --git a/offload/unittests/OffloadAPI/CMakeLists.txt b/offload/unittests/OffloadAPI/CMakeLists.txt
index 292ea1eb4852f..4917aa7fd4881 100644
--- a/offload/unittests/OffloadAPI/CMakeLists.txt
+++ b/offload/unittests/OffloadAPI/CMakeLists.txt
@@ -52,6 +52,7 @@ add_offload_unittest("program"
program/olDestroyProgram.cpp)
add_offload_unittest("queue"
+ queue/AMDGPUQueueUserCountTest.cpp
queue/olCreateQueue.cpp
queue/olSyncQueue.cpp
queue/olDestroyQueue.cpp
@@ -60,6 +61,8 @@ add_offload_unittest("queue"
queue/olWaitEvents.cpp
queue/olLaunchHostFunction.cpp
queue/olQueryQueue.cpp)
+target_include_directories("queue.unittests" PRIVATE
+ ${CMAKE_CURRENT_SOURCE_DIR}/../../plugins-nextgen/amdgpu/utils)
add_offload_unittest("symbol"
symbol/olGetSymbol.cpp
diff --git a/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp b/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
new file mode 100644
index 0000000000000..a04e2b701d0aa
--- /dev/null
+++ b/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
@@ -0,0 +1,44 @@
+//===------- Offload tests - AMDGPU queue user count ----------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPUQueueUserCount.h"
+#include <cstdint>
+#include <gtest/gtest.h>
+
+using llvm::omp::target::plugin::AMDGPUQueueUserCount;
+
+TEST(AMDGPUQueueUserCount, StartsAtZero) {
+ AMDGPUQueueUserCount Users;
+ EXPECT_EQ(Users.getUserCount(), 0u);
+}
+
+TEST(AMDGPUQueueUserCount, AddAndRemoveTrackIntegerCount) {
+ AMDGPUQueueUserCount Users;
+ Users.addUser();
+ Users.addUser();
+ Users.addUser();
+ EXPECT_EQ(Users.getUserCount(), 3u);
+
+ Users.removeUser();
+ EXPECT_EQ(Users.getUserCount(), 2u);
+}
+
+// assignNextQueue picks the least-used busy queue by comparing getUserCount().
+// A boolean conversion would collapse every non-zero count to 1, so 5 > 1
+// would be false.
+TEST(AMDGPUQueueUserCount, BusyQueuesCompareByIntegerCount) {
+ AMDGPUQueueUserCount Heavy;
+ AMDGPUQueueUserCount Light;
+ for (uint32_t I = 0; I < 5; ++I)
+ Heavy.addUser();
+ Light.addUser();
+
+ EXPECT_EQ(Heavy.getUserCount(), 5u);
+ EXPECT_EQ(Light.getUserCount(), 1u);
+ EXPECT_GT(Heavy.getUserCount(), Light.getUserCount());
+}
More information about the llvm-commits
mailing list