[llvm] [Offload] Return an integer user count for AMDGPU HSA queues (PR #223335)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 23 23:50:50 PDT 2026


https://github.com/StevenYangCC updated https://github.com/llvm/llvm-project/pull/223335

>From bff758deabab7fdea5d0cfc18d178ad31d8fd6fc Mon Sep 17 00:00:00 2001
From: "chengcang.yang" <yangchengcang at gmail.com>
Date: Mon, 14 Sep 2026 17:08:33 +0800
Subject: [PATCH 1/3] [Offload] Return an integer user count for AMDGPU HSA
 queues

getUserCount returned bool while the stored count is uint32_t, so
busy-queue selection treated every non-zero count as one. Return the
integer count so least-used assignment can distinguish differently
loaded queues.
---
 offload/plugins-nextgen/amdgpu/src/rtl.cpp    | 15 ++++---
 .../amdgpu/utils/AMDGPUQueueUserCount.h       | 34 ++++++++++++++
 offload/unittests/OffloadAPI/CMakeLists.txt   |  3 ++
 .../queue/AMDGPUQueueUserCountTest.cpp        | 44 +++++++++++++++++++
 4 files changed, 89 insertions(+), 7 deletions(-)
 create mode 100644 offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h
 create mode 100644 offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp

diff --git a/offload/plugins-nextgen/amdgpu/src/rtl.cpp b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
index 92afb883840ef..fb9015bf593f9 100644
--- a/offload/plugins-nextgen/amdgpu/src/rtl.cpp
+++ b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
@@ -30,6 +30,7 @@
 #include "Shared/Utils.h"
 #include "Utils/ELF.h"
 
+#include "AMDGPUQueueUserCount.h"
 #include "GlobalHandler.h"
 #include "OffloadAPI.h"
 #include "OpenMP/OMPT/Callback.h"
@@ -780,7 +781,7 @@ using AMDGPUSignalManagerTy = GenericDeviceResourceManagerTy<AMDGPUSignalRef>;
 /// Class holding an HSA queue to submit kernel and barrier packets.
 struct AMDGPUQueueTy {
   /// Create an empty queue.
-  AMDGPUQueueTy() : Queue(nullptr), Mutex(), NumUsers(0) {}
+  AMDGPUQueueTy() : Queue(nullptr), Mutex() {}
 
   /// Lazily initialize a new queue belonging to a specific agent.
   Error init(GenericDeviceTy &Device, hsa_agent_t Agent, int32_t QueueSize) {
@@ -813,17 +814,17 @@ struct AMDGPUQueueTy {
     return Plugin::check(Status, "error in hsa_queue_destroy: %s");
   }
 
-  /// Returns the number of streams, this queue is currently assigned to.
-  bool getUserCount() const { return NumUsers; }
+  /// Returns the number of streams this queue is currently assigned to.
+  uint32_t getUserCount() const { return UserCount.getUserCount(); }
 
   /// Returns if the underlying HSA queue is initialized.
   bool isInitialized() { return Queue != nullptr; }
 
   /// Decrement user count of the queue object.
-  void removeUser() { --NumUsers; }
+  void removeUser() { UserCount.removeUser(); }
 
   /// Increase user count of the queue object.
-  void addUser() { ++NumUsers; }
+  void addUser() { UserCount.addUser(); }
 
   /// Push a kernel launch to the queue. The kernel launch requires an output
   /// signal and can define an optional input signal (nullptr if none).
@@ -1006,9 +1007,9 @@ struct AMDGPUQueueTy {
   /// atomic operations. We can further investigate it if this is a bottleneck.
   std::mutex Mutex;
 
-  /// The number of streams, this queue is currently assigned to. A queue is
+  /// Tracks how many streams this queue is currently assigned to. A queue is
   /// considered idle when this is zero, otherwise: busy.
-  uint32_t NumUsers;
+  AMDGPUQueueUserCount UserCount;
 };
 
 /// Struct that implements a stream of asynchronous operations for AMDGPU
diff --git a/offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h b/offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h
new file mode 100644
index 0000000000000..67428ce60155a
--- /dev/null
+++ b/offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h
@@ -0,0 +1,34 @@
+//===- AMDGPUQueueUserCount.h - HSA queue stream user count -----*- C++ -*-===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#ifndef OFFLOAD_PLUGINS_NEXTGEN_AMDGPU_UTILS_AMDGPUQUEUEUSERCOUNT_H
+#define OFFLOAD_PLUGINS_NEXTGEN_AMDGPU_UTILS_AMDGPUQUEUEUSERCOUNT_H
+
+#include <cstdint>
+
+namespace llvm {
+namespace omp {
+namespace target {
+namespace plugin {
+
+/// Tracks how many streams currently use an HSA queue.
+struct AMDGPUQueueUserCount {
+  uint32_t getUserCount() const { return NumUsers; }
+  void addUser() { ++NumUsers; }
+  void removeUser() { --NumUsers; }
+
+private:
+  uint32_t NumUsers = 0;
+};
+
+} // namespace plugin
+} // namespace target
+} // namespace omp
+} // namespace llvm
+
+#endif // OFFLOAD_PLUGINS_NEXTGEN_AMDGPU_UTILS_AMDGPUQUEUEUSERCOUNT_H
diff --git a/offload/unittests/OffloadAPI/CMakeLists.txt b/offload/unittests/OffloadAPI/CMakeLists.txt
index 292ea1eb4852f..4917aa7fd4881 100644
--- a/offload/unittests/OffloadAPI/CMakeLists.txt
+++ b/offload/unittests/OffloadAPI/CMakeLists.txt
@@ -52,6 +52,7 @@ add_offload_unittest("program"
     program/olDestroyProgram.cpp)
 
 add_offload_unittest("queue"
+    queue/AMDGPUQueueUserCountTest.cpp
     queue/olCreateQueue.cpp
     queue/olSyncQueue.cpp
     queue/olDestroyQueue.cpp
@@ -60,6 +61,8 @@ add_offload_unittest("queue"
     queue/olWaitEvents.cpp
     queue/olLaunchHostFunction.cpp
     queue/olQueryQueue.cpp)
+target_include_directories("queue.unittests" PRIVATE
+    ${CMAKE_CURRENT_SOURCE_DIR}/../../plugins-nextgen/amdgpu/utils)
 
 add_offload_unittest("symbol"
     symbol/olGetSymbol.cpp
diff --git a/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp b/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
new file mode 100644
index 0000000000000..a04e2b701d0aa
--- /dev/null
+++ b/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
@@ -0,0 +1,44 @@
+//===------- Offload tests - AMDGPU queue user count ----------------------===//
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+//===----------------------------------------------------------------------===//
+
+#include "AMDGPUQueueUserCount.h"
+#include <cstdint>
+#include <gtest/gtest.h>
+
+using llvm::omp::target::plugin::AMDGPUQueueUserCount;
+
+TEST(AMDGPUQueueUserCount, StartsAtZero) {
+  AMDGPUQueueUserCount Users;
+  EXPECT_EQ(Users.getUserCount(), 0u);
+}
+
+TEST(AMDGPUQueueUserCount, AddAndRemoveTrackIntegerCount) {
+  AMDGPUQueueUserCount Users;
+  Users.addUser();
+  Users.addUser();
+  Users.addUser();
+  EXPECT_EQ(Users.getUserCount(), 3u);
+
+  Users.removeUser();
+  EXPECT_EQ(Users.getUserCount(), 2u);
+}
+
+// assignNextQueue picks the least-used busy queue by comparing getUserCount().
+// A boolean conversion would collapse every non-zero count to 1, so 5 > 1
+// would be false.
+TEST(AMDGPUQueueUserCount, BusyQueuesCompareByIntegerCount) {
+  AMDGPUQueueUserCount Heavy;
+  AMDGPUQueueUserCount Light;
+  for (uint32_t I = 0; I < 5; ++I)
+    Heavy.addUser();
+  Light.addUser();
+
+  EXPECT_EQ(Heavy.getUserCount(), 5u);
+  EXPECT_EQ(Light.getUserCount(), 1u);
+  EXPECT_GT(Heavy.getUserCount(), Light.getUserCount());
+}

>From 222a32b5c480859e41a902bac9aa4ab565fff001 Mon Sep 17 00:00:00 2001
From: "chengcang.yang" <yangchengcang at gmail.com>
Date: Thu, 24 Sep 2026 14:47:34 +0800
Subject: [PATCH 2/3] [Offload] Keep the AMDGPU queue user count on the
 existing queue type

getUserCount still returns the integer count, but that count stays in
AMDGPUQueueTy. A separate helper type is not required for the fix.
---
 offload/plugins-nextgen/amdgpu/src/rtl.cpp    | 13 ++++---
 .../amdgpu/utils/AMDGPUQueueUserCount.h       | 34 -------------------
 offload/unittests/OffloadAPI/CMakeLists.txt   |  2 --
 .../queue/AMDGPUQueueUserCountTest.cpp        | 21 ++++++++----
 4 files changed, 21 insertions(+), 49 deletions(-)
 delete mode 100644 offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h

diff --git a/offload/plugins-nextgen/amdgpu/src/rtl.cpp b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
index fb9015bf593f9..6d481a270c350 100644
--- a/offload/plugins-nextgen/amdgpu/src/rtl.cpp
+++ b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
@@ -30,7 +30,6 @@
 #include "Shared/Utils.h"
 #include "Utils/ELF.h"
 
-#include "AMDGPUQueueUserCount.h"
 #include "GlobalHandler.h"
 #include "OffloadAPI.h"
 #include "OpenMP/OMPT/Callback.h"
@@ -781,7 +780,7 @@ using AMDGPUSignalManagerTy = GenericDeviceResourceManagerTy<AMDGPUSignalRef>;
 /// Class holding an HSA queue to submit kernel and barrier packets.
 struct AMDGPUQueueTy {
   /// Create an empty queue.
-  AMDGPUQueueTy() : Queue(nullptr), Mutex() {}
+  AMDGPUQueueTy() : Queue(nullptr), Mutex(), NumUsers(0) {}
 
   /// Lazily initialize a new queue belonging to a specific agent.
   Error init(GenericDeviceTy &Device, hsa_agent_t Agent, int32_t QueueSize) {
@@ -815,16 +814,16 @@ struct AMDGPUQueueTy {
   }
 
   /// Returns the number of streams this queue is currently assigned to.
-  uint32_t getUserCount() const { return UserCount.getUserCount(); }
+  uint32_t getUserCount() const { return NumUsers; }
 
   /// Returns if the underlying HSA queue is initialized.
   bool isInitialized() { return Queue != nullptr; }
 
   /// Decrement user count of the queue object.
-  void removeUser() { UserCount.removeUser(); }
+  void removeUser() { --NumUsers; }
 
   /// Increase user count of the queue object.
-  void addUser() { UserCount.addUser(); }
+  void addUser() { ++NumUsers; }
 
   /// Push a kernel launch to the queue. The kernel launch requires an output
   /// signal and can define an optional input signal (nullptr if none).
@@ -1007,9 +1006,9 @@ struct AMDGPUQueueTy {
   /// atomic operations. We can further investigate it if this is a bottleneck.
   std::mutex Mutex;
 
-  /// Tracks how many streams this queue is currently assigned to. A queue is
+  /// The number of streams this queue is currently assigned to. A queue is
   /// considered idle when this is zero, otherwise: busy.
-  AMDGPUQueueUserCount UserCount;
+  uint32_t NumUsers;
 };
 
 /// Struct that implements a stream of asynchronous operations for AMDGPU
diff --git a/offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h b/offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h
deleted file mode 100644
index 67428ce60155a..0000000000000
--- a/offload/plugins-nextgen/amdgpu/utils/AMDGPUQueueUserCount.h
+++ /dev/null
@@ -1,34 +0,0 @@
-//===- AMDGPUQueueUserCount.h - HSA queue stream user count -----*- C++ -*-===//
-//
-// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
-// See https://llvm.org/LICENSE.txt for license information.
-// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
-//
-//===----------------------------------------------------------------------===//
-
-#ifndef OFFLOAD_PLUGINS_NEXTGEN_AMDGPU_UTILS_AMDGPUQUEUEUSERCOUNT_H
-#define OFFLOAD_PLUGINS_NEXTGEN_AMDGPU_UTILS_AMDGPUQUEUEUSERCOUNT_H
-
-#include <cstdint>
-
-namespace llvm {
-namespace omp {
-namespace target {
-namespace plugin {
-
-/// Tracks how many streams currently use an HSA queue.
-struct AMDGPUQueueUserCount {
-  uint32_t getUserCount() const { return NumUsers; }
-  void addUser() { ++NumUsers; }
-  void removeUser() { --NumUsers; }
-
-private:
-  uint32_t NumUsers = 0;
-};
-
-} // namespace plugin
-} // namespace target
-} // namespace omp
-} // namespace llvm
-
-#endif // OFFLOAD_PLUGINS_NEXTGEN_AMDGPU_UTILS_AMDGPUQUEUEUSERCOUNT_H
diff --git a/offload/unittests/OffloadAPI/CMakeLists.txt b/offload/unittests/OffloadAPI/CMakeLists.txt
index 4917aa7fd4881..2ee4e45105093 100644
--- a/offload/unittests/OffloadAPI/CMakeLists.txt
+++ b/offload/unittests/OffloadAPI/CMakeLists.txt
@@ -61,8 +61,6 @@ add_offload_unittest("queue"
     queue/olWaitEvents.cpp
     queue/olLaunchHostFunction.cpp
     queue/olQueryQueue.cpp)
-target_include_directories("queue.unittests" PRIVATE
-    ${CMAKE_CURRENT_SOURCE_DIR}/../../plugins-nextgen/amdgpu/utils)
 
 add_offload_unittest("symbol"
     symbol/olGetSymbol.cpp
diff --git a/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp b/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
index a04e2b701d0aa..986ab5d0e7d2d 100644
--- a/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
+++ b/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
@@ -6,19 +6,28 @@
 //
 //===----------------------------------------------------------------------===//
 
-#include "AMDGPUQueueUserCount.h"
 #include <cstdint>
 #include <gtest/gtest.h>
 
-using llvm::omp::target::plugin::AMDGPUQueueUserCount;
+// AMDGPUQueueTy lives in the AMDGPU plugin and is not separately linkable.
+// These are the same user-count operations: getUserCount returns the integer
+// count, and assignNextQueue compares those counts to pick the least-used
+// queue.
+struct QueueUserCount {
+  uint32_t getUserCount() const { return NumUsers; }
+  void addUser() { ++NumUsers; }
+  void removeUser() { --NumUsers; }
+
+  uint32_t NumUsers = 0;
+};
 
 TEST(AMDGPUQueueUserCount, StartsAtZero) {
-  AMDGPUQueueUserCount Users;
+  QueueUserCount Users;
   EXPECT_EQ(Users.getUserCount(), 0u);
 }
 
 TEST(AMDGPUQueueUserCount, AddAndRemoveTrackIntegerCount) {
-  AMDGPUQueueUserCount Users;
+  QueueUserCount Users;
   Users.addUser();
   Users.addUser();
   Users.addUser();
@@ -32,8 +41,8 @@ TEST(AMDGPUQueueUserCount, AddAndRemoveTrackIntegerCount) {
 // A boolean conversion would collapse every non-zero count to 1, so 5 > 1
 // would be false.
 TEST(AMDGPUQueueUserCount, BusyQueuesCompareByIntegerCount) {
-  AMDGPUQueueUserCount Heavy;
-  AMDGPUQueueUserCount Light;
+  QueueUserCount Heavy;
+  QueueUserCount Light;
   for (uint32_t I = 0; I < 5; ++I)
     Heavy.addUser();
   Light.addUser();

>From 93cb50fce6540886db94e5845159816f784f48e2 Mon Sep 17 00:00:00 2001
From: "chengcang.yang" <yangchengcang at gmail.com>
Date: Thu, 24 Sep 2026 14:50:05 +0800
Subject: [PATCH 3/3] [Offload] Drop the extra AMDGPU queue user count tests

The return type change is the fix. The helper type and the unit tests
are not needed.
---
 offload/plugins-nextgen/amdgpu/src/rtl.cpp    |  4 +-
 offload/unittests/OffloadAPI/CMakeLists.txt   |  1 -
 .../queue/AMDGPUQueueUserCountTest.cpp        | 53 -------------------
 3 files changed, 2 insertions(+), 56 deletions(-)
 delete mode 100644 offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp

diff --git a/offload/plugins-nextgen/amdgpu/src/rtl.cpp b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
index 6d481a270c350..8169231a5ceaa 100644
--- a/offload/plugins-nextgen/amdgpu/src/rtl.cpp
+++ b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
@@ -813,7 +813,7 @@ struct AMDGPUQueueTy {
     return Plugin::check(Status, "error in hsa_queue_destroy: %s");
   }
 
-  /// Returns the number of streams this queue is currently assigned to.
+  /// Returns the number of streams, this queue is currently assigned to.
   uint32_t getUserCount() const { return NumUsers; }
 
   /// Returns if the underlying HSA queue is initialized.
@@ -1006,7 +1006,7 @@ struct AMDGPUQueueTy {
   /// atomic operations. We can further investigate it if this is a bottleneck.
   std::mutex Mutex;
 
-  /// The number of streams this queue is currently assigned to. A queue is
+  /// The number of streams, this queue is currently assigned to. A queue is
   /// considered idle when this is zero, otherwise: busy.
   uint32_t NumUsers;
 };
diff --git a/offload/unittests/OffloadAPI/CMakeLists.txt b/offload/unittests/OffloadAPI/CMakeLists.txt
index 2ee4e45105093..292ea1eb4852f 100644
--- a/offload/unittests/OffloadAPI/CMakeLists.txt
+++ b/offload/unittests/OffloadAPI/CMakeLists.txt
@@ -52,7 +52,6 @@ add_offload_unittest("program"
     program/olDestroyProgram.cpp)
 
 add_offload_unittest("queue"
-    queue/AMDGPUQueueUserCountTest.cpp
     queue/olCreateQueue.cpp
     queue/olSyncQueue.cpp
     queue/olDestroyQueue.cpp
diff --git a/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp b/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
deleted file mode 100644
index 986ab5d0e7d2d..0000000000000
--- a/offload/unittests/OffloadAPI/queue/AMDGPUQueueUserCountTest.cpp
+++ /dev/null
@@ -1,53 +0,0 @@
-//===------- Offload tests - AMDGPU queue user count ----------------------===//
-//
-// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
-// See https://llvm.org/LICENSE.txt for license information.
-// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
-//
-//===----------------------------------------------------------------------===//
-
-#include <cstdint>
-#include <gtest/gtest.h>
-
-// AMDGPUQueueTy lives in the AMDGPU plugin and is not separately linkable.
-// These are the same user-count operations: getUserCount returns the integer
-// count, and assignNextQueue compares those counts to pick the least-used
-// queue.
-struct QueueUserCount {
-  uint32_t getUserCount() const { return NumUsers; }
-  void addUser() { ++NumUsers; }
-  void removeUser() { --NumUsers; }
-
-  uint32_t NumUsers = 0;
-};
-
-TEST(AMDGPUQueueUserCount, StartsAtZero) {
-  QueueUserCount Users;
-  EXPECT_EQ(Users.getUserCount(), 0u);
-}
-
-TEST(AMDGPUQueueUserCount, AddAndRemoveTrackIntegerCount) {
-  QueueUserCount Users;
-  Users.addUser();
-  Users.addUser();
-  Users.addUser();
-  EXPECT_EQ(Users.getUserCount(), 3u);
-
-  Users.removeUser();
-  EXPECT_EQ(Users.getUserCount(), 2u);
-}
-
-// assignNextQueue picks the least-used busy queue by comparing getUserCount().
-// A boolean conversion would collapse every non-zero count to 1, so 5 > 1
-// would be false.
-TEST(AMDGPUQueueUserCount, BusyQueuesCompareByIntegerCount) {
-  QueueUserCount Heavy;
-  QueueUserCount Light;
-  for (uint32_t I = 0; I < 5; ++I)
-    Heavy.addUser();
-  Light.addUser();
-
-  EXPECT_EQ(Heavy.getUserCount(), 5u);
-  EXPECT_EQ(Light.getUserCount(), 1u);
-  EXPECT_GT(Heavy.getUserCount(), Light.getUserCount());
-}



More information about the llvm-commits mailing list