[llvm] [Offload] Check Expected from hasPendingWorkImpl in AMDGPU dataFill (PR #223329)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 14 23:26:21 PDT 2026
https://github.com/StevenYangCC updated https://github.com/llvm/llvm-project/pull/223329
>From db1f6572d740a8bbf4d0cf173acc27b9de47a4f4 Mon Sep 17 00:00:00 2001
From: "chengcang.yang" <yangchengcang at gmail.com>
Date: Mon, 14 Sep 2026 16:25:57 +0800
Subject: [PATCH] [Offload] Check Expected from hasPendingWorkImpl in AMDGPU
dataFill
hasPendingWorkImpl returns Expected<bool>. Using that result directly as
a condition tests whether a value is present, not whether the queue has
pending work. The asynchronous fill path ran on every successful query,
and a failed query left an Error unconsumed.
Unwrap the Error first, then test the boolean. Add olMemFill tests for
the idle-queue and pending-work four-byte fill paths on AMDGPU.
---
offload/plugins-nextgen/amdgpu/src/rtl.cpp | 6 +-
.../unittests/OffloadAPI/memory/olMemFill.cpp | 57 +++++++++++++++++++
2 files changed, 62 insertions(+), 1 deletion(-)
diff --git a/offload/plugins-nextgen/amdgpu/src/rtl.cpp b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
index 281b9e3795a54..7a875e6cf9d8f 100644
--- a/offload/plugins-nextgen/amdgpu/src/rtl.cpp
+++ b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
@@ -3050,7 +3050,11 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
llvm_unreachable("Invalid pattern size");
}
- if (hasPendingWorkImpl(AsyncInfoWrapper)) {
+ auto Pending = hasPendingWorkImpl(AsyncInfoWrapper);
+ if (auto Err = Pending.takeError())
+ return Err;
+
+ if (*Pending) {
AMDGPUStreamTy *Stream = nullptr;
if (auto Err = getStream(AsyncInfoWrapper, Stream))
return Err;
diff --git a/offload/unittests/OffloadAPI/memory/olMemFill.cpp b/offload/unittests/OffloadAPI/memory/olMemFill.cpp
index 467a551c48c94..2565c910f02d1 100644
--- a/offload/unittests/OffloadAPI/memory/olMemFill.cpp
+++ b/offload/unittests/OffloadAPI/memory/olMemFill.cpp
@@ -9,6 +9,7 @@
#include "../common/Fixtures.hpp"
#include <OffloadAPI.h>
#include <array>
+#include <cstring>
#include <gtest/gtest.h>
#include <vector>
@@ -75,6 +76,62 @@ TEST_P(olMemFillTest, Success32Enqueue) {
test_body<uint32_t, 0xDEADBEEF, 1024, true>();
}
+// The AMDGPU plugin chooses a synchronous HSA fill when the queue is idle and
+// a host-callback fill when work is already pending. Those two paths used to
+// be collapsed because Expected<bool> was tested for "has a value" instead of
+// "has pending work".
+TEST_P(olMemFillTest, Success32IdleQueueCompletesWithoutSync) {
+ if (getPlatformBackend() != OL_PLATFORM_BACKEND_AMDGPU)
+ GTEST_SKIP() << "Idle-queue synchronous fill is AMDGPU-specific";
+
+ constexpr size_t Size = 1024;
+ void *Alloc;
+ ASSERT_SUCCESS(olMemAlloc(Device, OL_ALLOC_TYPE_MANAGED, Size, &Alloc));
+
+ uint32_t Pattern = 0xDEADBEEF;
+ ASSERT_SUCCESS(olMemFill(Queue, Alloc, sizeof(Pattern), &Pattern, Size));
+
+ bool IsQueueWorkCompleted = false;
+ ASSERT_SUCCESS(olQueryQueue(Queue, &IsQueueWorkCompleted));
+ ASSERT_TRUE(IsQueueWorkCompleted);
+
+ auto *AllocPtr = reinterpret_cast<uint32_t *>(Alloc);
+ for (size_t i = 0; i < Size / sizeof(Pattern); ++i)
+ ASSERT_EQ(AllocPtr[i], Pattern);
+
+ olMemFree(Alloc);
+}
+
+TEST_P(olMemFillTest, Success32EnqueueStaysPendingUntilHostTask) {
+ if (getPlatformBackend() != OL_PLATFORM_BACKEND_AMDGPU)
+ GTEST_SKIP() << "Pending-work asynchronous fill is AMDGPU-specific";
+
+ ManuallyTriggeredTask Manual;
+ ASSERT_SUCCESS(Manual.enqueue(Queue));
+
+ constexpr size_t Size = 1024;
+ void *Alloc;
+ ASSERT_SUCCESS(olMemAlloc(Device, OL_ALLOC_TYPE_MANAGED, Size, &Alloc));
+ std::memset(Alloc, 0, Size);
+
+ uint32_t Pattern = 0xDEADBEEF;
+ ASSERT_SUCCESS(olMemFill(Queue, Alloc, sizeof(Pattern), &Pattern, Size));
+
+ bool IsQueueWorkCompleted = false;
+ ASSERT_SUCCESS(olQueryQueue(Queue, &IsQueueWorkCompleted));
+ ASSERT_FALSE(IsQueueWorkCompleted);
+
+ auto *AllocPtr = reinterpret_cast<uint32_t *>(Alloc);
+ ASSERT_EQ(AllocPtr[0], 0u);
+
+ ASSERT_SUCCESS(Manual.trigger());
+ ASSERT_SUCCESS(olSyncQueue(Queue));
+ for (size_t i = 0; i < Size / sizeof(Pattern); ++i)
+ ASSERT_EQ(AllocPtr[i], Pattern);
+
+ olMemFree(Alloc);
+}
+
TEST_P(olMemFillTest, SuccessLarge) {
constexpr size_t Size = 1024;
void *Alloc;
More information about the llvm-commits
mailing list