[llvm] [Offload] Check Expected from hasPendingWorkImpl in AMDGPU dataFill (PR #223329)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 14 23:26:21 PDT 2026


https://github.com/StevenYangCC updated https://github.com/llvm/llvm-project/pull/223329

>From db1f6572d740a8bbf4d0cf173acc27b9de47a4f4 Mon Sep 17 00:00:00 2001
From: "chengcang.yang" <yangchengcang at gmail.com>
Date: Mon, 14 Sep 2026 16:25:57 +0800
Subject: [PATCH] [Offload] Check Expected from hasPendingWorkImpl in AMDGPU
 dataFill

hasPendingWorkImpl returns Expected<bool>. Using that result directly as
a condition tests whether a value is present, not whether the queue has
pending work. The asynchronous fill path ran on every successful query,
and a failed query left an Error unconsumed.

Unwrap the Error first, then test the boolean. Add olMemFill tests for
the idle-queue and pending-work four-byte fill paths on AMDGPU.
---
 offload/plugins-nextgen/amdgpu/src/rtl.cpp    |  6 +-
 .../unittests/OffloadAPI/memory/olMemFill.cpp | 57 +++++++++++++++++++
 2 files changed, 62 insertions(+), 1 deletion(-)

diff --git a/offload/plugins-nextgen/amdgpu/src/rtl.cpp b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
index 281b9e3795a54..7a875e6cf9d8f 100644
--- a/offload/plugins-nextgen/amdgpu/src/rtl.cpp
+++ b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
@@ -3050,7 +3050,11 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
         llvm_unreachable("Invalid pattern size");
       }
 
-      if (hasPendingWorkImpl(AsyncInfoWrapper)) {
+      auto Pending = hasPendingWorkImpl(AsyncInfoWrapper);
+      if (auto Err = Pending.takeError())
+        return Err;
+
+      if (*Pending) {
         AMDGPUStreamTy *Stream = nullptr;
         if (auto Err = getStream(AsyncInfoWrapper, Stream))
           return Err;
diff --git a/offload/unittests/OffloadAPI/memory/olMemFill.cpp b/offload/unittests/OffloadAPI/memory/olMemFill.cpp
index 467a551c48c94..2565c910f02d1 100644
--- a/offload/unittests/OffloadAPI/memory/olMemFill.cpp
+++ b/offload/unittests/OffloadAPI/memory/olMemFill.cpp
@@ -9,6 +9,7 @@
 #include "../common/Fixtures.hpp"
 #include <OffloadAPI.h>
 #include <array>
+#include <cstring>
 #include <gtest/gtest.h>
 #include <vector>
 
@@ -75,6 +76,62 @@ TEST_P(olMemFillTest, Success32Enqueue) {
   test_body<uint32_t, 0xDEADBEEF, 1024, true>();
 }
 
+// The AMDGPU plugin chooses a synchronous HSA fill when the queue is idle and
+// a host-callback fill when work is already pending. Those two paths used to
+// be collapsed because Expected<bool> was tested for "has a value" instead of
+// "has pending work".
+TEST_P(olMemFillTest, Success32IdleQueueCompletesWithoutSync) {
+  if (getPlatformBackend() != OL_PLATFORM_BACKEND_AMDGPU)
+    GTEST_SKIP() << "Idle-queue synchronous fill is AMDGPU-specific";
+
+  constexpr size_t Size = 1024;
+  void *Alloc;
+  ASSERT_SUCCESS(olMemAlloc(Device, OL_ALLOC_TYPE_MANAGED, Size, &Alloc));
+
+  uint32_t Pattern = 0xDEADBEEF;
+  ASSERT_SUCCESS(olMemFill(Queue, Alloc, sizeof(Pattern), &Pattern, Size));
+
+  bool IsQueueWorkCompleted = false;
+  ASSERT_SUCCESS(olQueryQueue(Queue, &IsQueueWorkCompleted));
+  ASSERT_TRUE(IsQueueWorkCompleted);
+
+  auto *AllocPtr = reinterpret_cast<uint32_t *>(Alloc);
+  for (size_t i = 0; i < Size / sizeof(Pattern); ++i)
+    ASSERT_EQ(AllocPtr[i], Pattern);
+
+  olMemFree(Alloc);
+}
+
+TEST_P(olMemFillTest, Success32EnqueueStaysPendingUntilHostTask) {
+  if (getPlatformBackend() != OL_PLATFORM_BACKEND_AMDGPU)
+    GTEST_SKIP() << "Pending-work asynchronous fill is AMDGPU-specific";
+
+  ManuallyTriggeredTask Manual;
+  ASSERT_SUCCESS(Manual.enqueue(Queue));
+
+  constexpr size_t Size = 1024;
+  void *Alloc;
+  ASSERT_SUCCESS(olMemAlloc(Device, OL_ALLOC_TYPE_MANAGED, Size, &Alloc));
+  std::memset(Alloc, 0, Size);
+
+  uint32_t Pattern = 0xDEADBEEF;
+  ASSERT_SUCCESS(olMemFill(Queue, Alloc, sizeof(Pattern), &Pattern, Size));
+
+  bool IsQueueWorkCompleted = false;
+  ASSERT_SUCCESS(olQueryQueue(Queue, &IsQueueWorkCompleted));
+  ASSERT_FALSE(IsQueueWorkCompleted);
+
+  auto *AllocPtr = reinterpret_cast<uint32_t *>(Alloc);
+  ASSERT_EQ(AllocPtr[0], 0u);
+
+  ASSERT_SUCCESS(Manual.trigger());
+  ASSERT_SUCCESS(olSyncQueue(Queue));
+  for (size_t i = 0; i < Size / sizeof(Pattern); ++i)
+    ASSERT_EQ(AllocPtr[i], Pattern);
+
+  olMemFree(Alloc);
+}
+
 TEST_P(olMemFillTest, SuccessLarge) {
   constexpr size_t Size = 1024;
   void *Alloc;



More information about the llvm-commits mailing list