[llvm-branch-commits] [llvm] [Offload][AMDGPU] Wire HSA profiling into GenericProfiler abstraction (PR #226044)
Jan Patrick Lehr via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Thu Sep 24 00:18:07 PDT 2026
https://github.com/jplehr created https://github.com/llvm/llvm-project/pull/226044
Add device profiling infrastructure to the AMDGPU plugin so that the
GenericProfiler can receive nanosecond-accurate kernel execution and
data transfer timestamps from the HSA runtime.
Key changes:
- Add ProfilingInfoTy struct to transport HSA profiling data
- Add timeKernelInNsAsync/timeDataTransferInNsAsync callbacks that
extract dispatch/copy times from HSA signals and call
handleKernelCompletion/handleDataTransfer on the profiler
- Add getOrNullProfilerSpecificData helper to extract ProfilerData
from AsyncInfoWrapperTy
- Add getDeviceTimeStamp() override using hsa_system_get_info
- Add getSystemTimestampInNs() for HSA system timestamp queries
- Add schedProfilerKernelTiming/schedProfilerDataTransferTiming to
StreamSlotTy for scheduling profiler callbacks on stream slots
- Thread ProfilerSpecificData through pushKernelLaunch,
pushMemoryCopyH2DAsync, pushMemoryCopyD2HAsync, pushMemoryCopyD2DAsync
- Extract ProfilerSpecificData in dataSubmitImpl, dataRetrieveImpl,
dataExchangeImpl, and launchImpl
- Add stub getDeviceTimeStamp() to CUDA plugin
Assisted-by: Cursor
Assisted-by: Claude Code
---
<sub>Stack created with <a href="https://github.com/github/gh-stack">GitHub Stacks CLI</a> • <a href="https://gh.io/stacks-feedback">Give Feedback 💬</a></sub>
>From 166477014dd6c42c4fcfa83a7193504dcddb9e93 Mon Sep 17 00:00:00 2001
From: JP Lehr <JanPatrick.Lehr at amd.com>
Date: Thu, 2 Apr 2026 07:36:31 -0500
Subject: [PATCH] [Offload][AMDGPU] Wire HSA profiling into GenericProfiler
abstraction
Add device profiling infrastructure to the AMDGPU plugin so that the
GenericProfiler can receive nanosecond-accurate kernel execution and
data transfer timestamps from the HSA runtime.
Key changes:
- Add ProfilingInfoTy struct to transport HSA profiling data
- Add timeKernelInNsAsync/timeDataTransferInNsAsync callbacks that
extract dispatch/copy times from HSA signals and call
handleKernelCompletion/handleDataTransfer on the profiler
- Add getOrNullProfilerSpecificData helper to extract ProfilerData
from AsyncInfoWrapperTy
- Add getDeviceTimeStamp() override using hsa_system_get_info
- Add getSystemTimestampInNs() for HSA system timestamp queries
- Add schedProfilerKernelTiming/schedProfilerDataTransferTiming to
StreamSlotTy for scheduling profiler callbacks on stream slots
- Thread ProfilerSpecificData through pushKernelLaunch,
pushMemoryCopyH2DAsync, pushMemoryCopyD2HAsync, pushMemoryCopyD2DAsync
- Extract ProfilerSpecificData in dataSubmitImpl, dataRetrieveImpl,
dataExchangeImpl, and launchImpl
- Add stub getDeviceTimeStamp() to CUDA plugin
Assisted-by: Cursor
Assisted-by: Claude Code
---
.../amdgpu/dynamic_hsa/hsa.cpp | 1 +
.../amdgpu/dynamic_hsa/hsa_ext_amd.h | 8 +
offload/plugins-nextgen/amdgpu/src/rtl.cpp | 235 ++++++++++++++++--
.../common/include/PluginInterface.h | 21 +-
.../common/src/PluginInterface.cpp | 24 +-
offload/plugins-nextgen/cuda/src/rtl.cpp | 24 +-
offload/plugins-nextgen/host/src/rtl.cpp | 12 +-
.../level_zero/include/L0Device.h | 9 +-
.../level_zero/include/L0Kernel.h | 3 +-
.../level_zero/src/L0Device.cpp | 12 +-
.../level_zero/src/L0Kernel.cpp | 3 +-
11 files changed, 295 insertions(+), 57 deletions(-)
diff --git a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp
index 6cb81f06dd9c4..88273df086467 100644
--- a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp
+++ b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp
@@ -71,6 +71,7 @@ DLWRAP(hsa_amd_signal_create, 5)
DLWRAP(hsa_amd_signal_async_handler, 5)
DLWRAP(hsa_amd_pointer_info, 5)
DLWRAP(hsa_amd_profiling_get_dispatch_time, 3)
+DLWRAP(hsa_amd_profiling_get_async_copy_time, 2)
DLWRAP(hsa_amd_profiling_set_profiler_enabled, 2)
DLWRAP(hsa_code_object_reader_create_from_memory, 3)
DLWRAP(hsa_code_object_reader_destroy, 1)
diff --git a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h
index c736ec0759841..9a00d293bbef6 100644
--- a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h
+++ b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h
@@ -208,6 +208,14 @@ hsa_amd_profiling_get_dispatch_time(hsa_agent_t agent, hsa_signal_t signal,
hsa_status_t hsa_amd_profiling_set_profiler_enabled(hsa_queue_t *queue,
int enable);
+typedef struct hsa_amd_profiling_async_copy_time_s {
+ uint64_t start;
+ uint64_t end;
+} hsa_amd_profiling_async_copy_time_t;
+
+hsa_status_t hsa_amd_profiling_get_async_copy_time(
+ hsa_signal_t signal, hsa_amd_profiling_async_copy_time_t *time);
+
hsa_status_t hsa_amd_vmem_address_reserve(void **va, size_t size,
uint64_t address, uint64_t flags);
diff --git a/offload/plugins-nextgen/amdgpu/src/rtl.cpp b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
index 888830f7dfa03..5fe8066f22836 100644
--- a/offload/plugins-nextgen/amdgpu/src/rtl.cpp
+++ b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
@@ -30,6 +30,7 @@
#include "Shared/Utils.h"
#include "Utils/ELF.h"
+#include "GenericProfiler.h"
#include "GlobalHandler.h"
#include "OffloadAPI.h"
#include "OpenMP/OMPT/Callback.h"
@@ -95,6 +96,89 @@ struct AMDGPUEventManagerTy;
struct AMDGPUDeviceImageTy;
struct AMDGPUMemoryManagerTy;
struct AMDGPUMemoryPoolTy;
+struct AMDGPUSignalTy;
+
+/// Use to transport information to profiler timing functions.
+/// The profiler is captured as a pointer because this outlives the call that
+/// schedules the asynchronous action.
+struct ProfilingInfoTy {
+ GenericProfilerTy *Profiler;
+ hsa_agent_t Agent;
+ AMDGPUSignalTy *Signal;
+ double TicksToTime;
+ void *ProfilerSpecificData;
+};
+
+static ProfilingInfoTy *getProfilingInfo(void *Data);
+
+static std::pair<uint64_t, uint64_t>
+getKernelStartAndEndTime(const ProfilingInfoTy *Args);
+
+static std::pair<uint64_t, uint64_t>
+getCopyStartAndEndTime(const ProfilingInfoTy *Args);
+
+static Error timeKernelInNsAsync(void *Data);
+
+static Error timeDataTransferInNsAsync(void *Data) {
+ auto Args = getProfilingInfo(Data);
+ auto [Start, End] = getCopyStartAndEndTime(Args);
+ Args->Profiler->handleDataTransfer(Start, End, Args->ProfilerSpecificData);
+ return Plugin::success();
+}
+
+static void *
+getOrNullProfilerSpecificData(AsyncInfoWrapperTy &AsyncInfoWrapper) {
+ __tgt_async_info *AI = AsyncInfoWrapper;
+ return AI ? AI->ProfilerData : nullptr;
+}
+
+} // namespace plugin
+} // namespace target
+} // namespace omp
+} // namespace llvm
+
+static double setTicksToTime() {
+ uint64_t TicksFrequency = 1;
+ double TicksToTime = 1.0;
+
+ hsa_status_t Status =
+ hsa_system_get_info(HSA_SYSTEM_INFO_TIMESTAMP_FREQUENCY, &TicksFrequency);
+ if (Status == HSA_STATUS_SUCCESS)
+ TicksToTime = (double)1e9 / (double)TicksFrequency;
+
+ return TicksToTime;
+}
+
+static double TicksToTime = 1.0;
+
+static void setHSATicksToTimeConstant() { TicksToTime = setTicksToTime(); }
+
+/// Get the current HSA-based system timestamp in nanoseconds.
+/// Called by OmptTracing.cpp (from PluginOmpt) for device time queries.
+uint64_t getSystemTimestampInNs() {
+ uint64_t TimeStamp = 0;
+ hsa_status_t Status =
+ hsa_system_get_info(HSA_SYSTEM_INFO_TIMESTAMP, &TimeStamp);
+ if (Status != HSA_STATUS_SUCCESS)
+ return 0;
+ return TimeStamp;
+}
+
+/// Enable or disable HSA async copy profiling for OMPT device tracing.
+/// Called by OmptTracing.cpp when a tool activates/deactivates device tracing.
+/// Full HSA profiling integration (enabling per-copy timing signals) will be
+/// wired up in a follow-on commit.
+void setOmptAsyncCopyProfile(bool Enable) {}
+
+/// Enable or disable HSA queue kernel profiling for OMPT device tracing.
+/// Called by OmptTracing.cpp when a tool activates/deactivates device tracing.
+/// Full HSA queue profiling integration will be wired up in a follow-on commit.
+void setGlobalOmptKernelProfile(void *Device, int Enable) {}
+
+namespace llvm {
+namespace omp {
+namespace target {
+namespace plugin {
namespace hsa_utils {
@@ -649,7 +733,8 @@ struct AMDGPUKernelTy : public GenericKernelTy {
Error launchImpl(GenericDeviceTy &GenericDevice, uint32_t NumThreads[3],
uint32_t NumBlocks[3], uint32_t DynBlockMemSize,
KernelLaunchArgsTy &LaunchArgs,
- AsyncInfoWrapperTy &AsyncInfoWrapper) const override;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) const override;
/// Return maximum block size for maximum occupancy
///
@@ -1067,6 +1152,7 @@ struct AMDGPUStreamTy {
MemcpyArgsTy MemcpyArgs;
ReleaseBufferArgsTy ReleaseBufferArgs;
ReleaseSignalArgsTy ReleaseSignalArgs;
+ ProfilingInfoTy ProfilerArgs;
void *CallbackArgs;
};
@@ -1104,7 +1190,31 @@ struct AMDGPUStreamTy {
return Plugin::success();
}
- /// Register a callback to be called on compleition
+ /// Schedule kernel timing measurement via the profiler on the slot.
+ Error schedProfilerKernelTiming(GenericProfilerTy *ProfilerPtr,
+ hsa_agent_t Agent,
+ AMDGPUSignalTy *OutputSignal,
+ double TicksToTime,
+ void *ProfilerSpecificData) {
+ Callbacks.emplace_back(timeKernelInNsAsync);
+ ActionArgs.emplace_back().ProfilerArgs = ProfilingInfoTy{
+ ProfilerPtr, Agent, OutputSignal, TicksToTime, ProfilerSpecificData};
+ return Plugin::success();
+ }
+
+ /// Schedule data transfer timing via the profiler on the slot.
+ Error schedProfilerDataTransferTiming(GenericProfilerTy *ProfilerPtr,
+ hsa_agent_t Agent,
+ AMDGPUSignalTy *OutputSignal,
+ double TicksToTime,
+ void *ProfilerSpecificData) {
+ Callbacks.emplace_back(timeDataTransferInNsAsync);
+ ActionArgs.emplace_back().ProfilerArgs = ProfilingInfoTy{
+ ProfilerPtr, Agent, OutputSignal, TicksToTime, ProfilerSpecificData};
+ return Plugin::success();
+ }
+
+ /// Register a callback to be called on completion.
Error schedCallback(AMDGPUStreamCallbackTy *Func, void *Data) {
Callbacks.emplace_back(Func);
ActionArgs.emplace_back().CallbackArgs = Data;
@@ -1129,6 +1239,12 @@ struct AMDGPUStreamTy {
} else if (Callback == releaseSignalAction) {
if (auto Err = releaseSignalAction(&ActionArg))
return Err;
+ } else if (Callback == timeKernelInNsAsync) {
+ if (auto Err = timeKernelInNsAsync(&ActionArg))
+ return Err;
+ } else if (Callback == timeDataTransferInNsAsync) {
+ if (auto Err = timeDataTransferInNsAsync(&ActionArg))
+ return Err;
} else if (Callback) {
if (auto Err = Callback(ActionArg.CallbackArgs))
return Err;
@@ -1400,7 +1516,9 @@ struct AMDGPUStreamTy {
Error pushKernelLaunch(const AMDGPUKernelTy &Kernel, void *KernelArgs,
uint32_t NumThreads[3], uint32_t NumBlocks[3],
uint32_t GroupSize, uint64_t StackSize,
- AMDGPUMemoryManagerTy &MemoryManager) {
+ AMDGPUMemoryManagerTy &MemoryManager,
+ void *ProfilerSpecificData = nullptr,
+ GenericProfilerTy *ProfilerPtr = nullptr) {
if (Queue == nullptr)
return Plugin::error(ErrorCode::INVALID_NULL_POINTER,
"target queue was nullptr");
@@ -1421,6 +1539,15 @@ struct AMDGPUStreamTy {
if (auto Err = Slots[Curr].schedReleaseBuffer(KernelArgs, MemoryManager))
return Err;
+#ifdef OMPT_SUPPORT
+ if (ProfilerSpecificData) {
+ if (auto Err = Slots[Curr].schedProfilerKernelTiming(
+ ProfilerPtr, Agent, OutputSignal, TicksToTime,
+ ProfilerSpecificData))
+ return Err;
+ }
+#endif
+
// If we are running an RPC server we want to wake up the server thread
// whenever there is a kernel running and let it sleep otherwise.
if (Device.getRPCServer())
@@ -1483,7 +1610,9 @@ struct AMDGPUStreamTy {
/// manager once the operation completes.
Error pushMemoryCopyD2HAsync(void *Dst, const void *Src, void *Inter,
uint64_t CopySize,
- AMDGPUMemoryManagerTy &MemoryManager) {
+ AMDGPUMemoryManagerTy &MemoryManager,
+ void *ProfilerSpecificData = nullptr,
+ GenericProfilerTy *ProfilerPtr = nullptr) {
// Retrieve available signals for the operation's outputs.
AMDGPUSignalTy *OutputSignals[2] = {};
if (auto Err = SignalManager.getResources(/*Num=*/2, OutputSignals))
@@ -1546,6 +1675,8 @@ struct AMDGPUStreamTy {
Error pushMemoryCopyH2DAsync(void *Dst, const void *Src, void *Inter,
uint64_t CopySize,
AMDGPUMemoryManagerTy &MemoryManager,
+ void *ProfilerSpecificData = nullptr,
+ GenericProfilerTy *ProfilerPtr = nullptr,
size_t NumTimes = 1) {
// Retrieve available signals for the operation's outputs.
AMDGPUSignalTy *OutputSignals[2] = {};
@@ -1621,7 +1752,9 @@ struct AMDGPUStreamTy {
// AMDGPUDeviceTy is incomplete here, passing the underlying agent instead
Error pushMemoryCopyD2DAsync(void *Dst, hsa_agent_t DstAgent, const void *Src,
- hsa_agent_t SrcAgent, uint64_t CopySize) {
+ hsa_agent_t SrcAgent, uint64_t CopySize,
+ void *ProfilerSpecificData = nullptr,
+ GenericProfilerTy *ProfilerPtr = nullptr) {
AMDGPUSignalTy *OutputSignal;
if (auto Err = SignalManager.getResources(/*Num=*/1, &OutputSignal))
return Err;
@@ -2288,6 +2421,8 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
if (auto Err = initMemoryPools())
return Err;
+ setHSATicksToTimeConstant();
+
char GPUName[64];
if (auto Err = getDeviceAttr(HSA_AGENT_INFO_NAME, GPUName))
return Err;
@@ -2632,6 +2767,9 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
/// Returns the clock frequency for the given AMDGPU device.
uint64_t getClockFrequency() const override { return ClockFrequency; }
+ /// Returns the current HSA system timestamp for profiling.
+ uint64_t getDeviceTimeStamp() override { return getSystemTimestampInNs(); }
+
/// Returns the HSA system timestamp frequency. Zero means unavailable.
uint64_t getSystemTimestampFrequency() const {
return SystemTimestampFrequency;
@@ -2843,10 +2981,13 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
/// Submit data to the device (host to device transfer).
Error dataSubmitImpl(void *TgtPtr, const void *HstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override {
AMDGPUStreamTy *Stream = nullptr;
void *PinnedPtr = nullptr;
+ auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
// Use one-step asynchronous operation when host memory is already pinned.
if (void *PinnedPtr =
PinnedAllocs.getDeviceAccessiblePtrFromPinnedBuffer(HstPtr)) {
@@ -2897,15 +3038,19 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
return Err;
return Stream->pushMemoryCopyH2DAsync(TgtPtr, HstPtr, PinnedPtr, Size,
- PinnedMemoryManager);
+ PinnedMemoryManager,
+ ProfilerSpecificData, ProfilerPtr);
}
/// Retrieve data from the device (device to host transfer).
Error dataRetrieveImpl(void *HstPtr, const void *TgtPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override {
AMDGPUStreamTy *Stream = nullptr;
void *PinnedPtr = nullptr;
+ auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
// Use one-step asynchronous operation when host memory is already pinned.
if (void *PinnedPtr =
PinnedAllocs.getDeviceAccessiblePtrFromPinnedBuffer(HstPtr)) {
@@ -2957,7 +3102,8 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
return Err;
return Stream->pushMemoryCopyD2HAsync(HstPtr, TgtPtr, PinnedPtr, Size,
- PinnedMemoryManager);
+ PinnedMemoryManager,
+ ProfilerSpecificData, ProfilerPtr);
}
Error dataMemcpyImpl(void *DstPtr, const void *SrcPtr, int64_t Size,
@@ -2990,9 +3136,12 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
/// Exchange data between two devices within the plugin.
Error dataExchangeImpl(const void *SrcPtr, GenericDeviceTy &DstGenericDevice,
void *DstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override {
AMDGPUDeviceTy &DstDevice = static_cast<AMDGPUDeviceTy &>(DstGenericDevice);
+ auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
// For large transfers use synchronous behavior.
if (Size >= OMPX_MaxAsyncCopyBytes) {
if (AsyncInfoWrapper.hasQueue())
@@ -3021,7 +3170,8 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
return Plugin::success();
return Stream->pushMemoryCopyD2DAsync(DstPtr, DstDevice.getAgent(), SrcPtr,
- getAgent(), (uint64_t)Size);
+ getAgent(), (uint64_t)Size,
+ ProfilerSpecificData, ProfilerPtr);
}
/// Insert a data fence between previous data operations and the following
@@ -3097,9 +3247,10 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
if (auto Err = getStream(AsyncInfoWrapper, Stream))
return Err;
- return Stream->pushMemoryCopyH2DAsync(TgtPtr, PatternPtr, PinnedPtr,
- PatternSize, PinnedMemoryManager,
- Size / PatternSize);
+ return Stream->pushMemoryCopyH2DAsync(
+ TgtPtr, PatternPtr, PinnedPtr, PatternSize, PinnedMemoryManager,
+ /*ProfilerSpecificData=*/nullptr, /*ProfilerPtr=*/nullptr,
+ Size / PatternSize);
}
interop_spec_t selectInteropPreference(int32_t InteropType,
@@ -3592,9 +3743,9 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
KernelLaunchArgsTy LaunchArgs = {};
uint32_t NumBlocksAndThreads[3] = {1u, 1u, 1u};
- auto Err =
- AMDGPUKernel.launchImpl(*this, NumBlocksAndThreads, NumBlocksAndThreads,
- 0, LaunchArgs, AsyncInfoWrapper);
+ auto Err = AMDGPUKernel.launchImpl(
+ *this, NumBlocksAndThreads, NumBlocksAndThreads, 0, LaunchArgs,
+ AsyncInfoWrapper, /*ProfilerPtr=*/nullptr);
AsyncInfoWrapper.finalize(Err);
return Err;
@@ -4370,7 +4521,8 @@ Error AMDGPUKernelTy::launchImpl(GenericDeviceTy &GenericDevice,
uint32_t NumThreads[3], uint32_t NumBlocks[3],
uint32_t DynBlockMemSize,
KernelLaunchArgsTy &LaunchArgs,
- AsyncInfoWrapperTy &AsyncInfoWrapper) const {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) const {
// Cooperative kernel launch is not yet supported for AMDGPU
if (LaunchArgs.Flags.Cooperative)
return Plugin::error(ErrorCode::UNSUPPORTED,
@@ -4455,10 +4607,12 @@ Error AMDGPUKernelTy::launchImpl(GenericDeviceTy &GenericDevice,
// HSA requires the group segment size to include both static and dynamic.
uint32_t TotalBlockMemSize = getStaticBlockMemSize() + DynBlockMemSize;
+ auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
// Push the kernel launch into the stream.
- return Stream->pushKernelLaunch(*this, AllArgs, NumThreads, NumBlocks,
- TotalBlockMemSize, StackSize,
- ArgsMemoryManager);
+ return Stream->pushKernelLaunch(
+ *this, AllArgs, NumThreads, NumBlocks, TotalBlockMemSize, StackSize,
+ ArgsMemoryManager, ProfilerSpecificData, ProfilerPtr);
}
Error AMDGPUKernelTy::printLaunchInfoDetails(
@@ -4649,6 +4803,45 @@ void AMDGPUQueueTy::callbackError(hsa_status_t Status, hsa_queue_t *Source,
FATAL_MESSAGE(1, "%s", toString(std::move(Err)).data());
}
+/// Implementation of profiling helper functions.
+static ProfilingInfoTy *getProfilingInfo(void *Data) {
+ return reinterpret_cast<ProfilingInfoTy *>(Data);
+}
+
+static std::pair<uint64_t, uint64_t>
+getKernelStartAndEndTime(const ProfilingInfoTy *Args) {
+ hsa_amd_profiling_dispatch_time_t Time = {};
+ hsa_status_t Status = hsa_amd_profiling_get_dispatch_time(
+ Args->Agent, Args->Signal->get(), &Time);
+ if (Status != HSA_STATUS_SUCCESS)
+ return {0, 0};
+ return {static_cast<uint64_t>(Time.start * Args->TicksToTime),
+ static_cast<uint64_t>(Time.end * Args->TicksToTime)};
+}
+
+static std::pair<uint64_t, uint64_t>
+getCopyStartAndEndTime(const ProfilingInfoTy *Args) {
+ hsa_amd_profiling_async_copy_time_t Time = {};
+ hsa_status_t Status =
+ hsa_amd_profiling_get_async_copy_time(Args->Signal->get(), &Time);
+ if (Status != HSA_STATUS_SUCCESS)
+ return {0, 0};
+ return {static_cast<uint64_t>(Time.start * Args->TicksToTime),
+ static_cast<uint64_t>(Time.end * Args->TicksToTime)};
+}
+
+static Error timeKernelInNsAsync(void *Data) {
+ assert(Data && "Invalid data pointer");
+ auto ProfilerInfo = getProfilingInfo(Data);
+ assert(ProfilerInfo && "Invalid profiling info");
+ assert(ProfilerInfo->ProfilerSpecificData && "Invalid ProfilerSpecificData");
+
+ auto [StartTime, EndTime] = getKernelStartAndEndTime(ProfilerInfo);
+ ProfilerInfo->Profiler->handleKernelCompletion(
+ StartTime, EndTime, ProfilerInfo->ProfilerSpecificData);
+ return Plugin::success();
+}
+
} // namespace plugin
} // namespace target
} // namespace omp
diff --git a/offload/plugins-nextgen/common/include/PluginInterface.h b/offload/plugins-nextgen/common/include/PluginInterface.h
index 17f698d0ed0a8..307198c315cb9 100644
--- a/offload/plugins-nextgen/common/include/PluginInterface.h
+++ b/offload/plugins-nextgen/common/include/PluginInterface.h
@@ -500,7 +500,8 @@ struct GenericKernelTy {
uint32_t NumThreads[3], uint32_t NumBlocks[3],
uint32_t DynBlockMemSize,
KernelLaunchArgsTy &LaunchArgs,
- AsyncInfoWrapperTy &AsyncInfoWrapper) const = 0;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr = nullptr) const = 0;
virtual Expected<uint64_t> maxGroupSize(GenericDeviceTy &GenericDevice,
uint64_t DynamicMemSize) const = 0;
@@ -1183,15 +1184,19 @@ struct GenericDeviceTy : public DeviceAllocatorTy {
/// Submit data to the device (host to device transfer).
Error dataSubmit(void *TgtPtr, const void *HstPtr, int64_t Size,
- __tgt_async_info *AsyncInfo);
+ __tgt_async_info *AsyncInfo,
+ GenericProfilerTy *ProfilerPtr = nullptr);
virtual Error dataSubmitImpl(void *TgtPtr, const void *HstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) = 0;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr = nullptr) = 0;
/// Retrieve data from the device (device to host transfer).
Error dataRetrieve(void *HstPtr, const void *TgtPtr, int64_t Size,
- __tgt_async_info *AsyncInfo);
+ __tgt_async_info *AsyncInfo,
+ GenericProfilerTy *ProfilerPtr = nullptr);
virtual Error dataRetrieveImpl(void *HstPtr, const void *TgtPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) = 0;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr = nullptr) = 0;
/// Copy data between arbitrary memory locations.
Error dataMemcpy(void *DstPtr, const void *SrcPtr, int64_t Size,
@@ -1207,10 +1212,12 @@ struct GenericDeviceTy : public DeviceAllocatorTy {
/// function is only valid if GenericPlugin::isDataExchangable() passing the
/// two devices returns true.
Error dataExchange(const void *SrcPtr, GenericDeviceTy &DstDev, void *DstPtr,
- int64_t Size, __tgt_async_info *AsyncInfo);
+ int64_t Size, __tgt_async_info *AsyncInfo,
+ GenericProfilerTy *ProfilerPtr = nullptr);
virtual Error dataExchangeImpl(const void *SrcPtr, GenericDeviceTy &DstDev,
void *DstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) = 0;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr = nullptr) = 0;
/// Fill data on the device with a pattern from the host
Error dataFill(void *TgtPtr, const void *PatternPtr, int64_t PatternSize,
diff --git a/offload/plugins-nextgen/common/src/PluginInterface.cpp b/offload/plugins-nextgen/common/src/PluginInterface.cpp
index 281ecc7753bcc..9b4dee9e19853 100644
--- a/offload/plugins-nextgen/common/src/PluginInterface.cpp
+++ b/offload/plugins-nextgen/common/src/PluginInterface.cpp
@@ -339,9 +339,9 @@ Error GenericKernelTy::launch(GenericDeviceTy &GenericDevice,
Profiler.handlePreKernelLaunch(&GenericDevice, EffectiveNumBlocks,
AsyncInfoWrapper);
- if (auto Err =
- launchImpl(GenericDevice, EffectiveNumThreads, EffectiveNumBlocks,
- DynBlockMemConf.NativeSize, LaunchArgs, AsyncInfoWrapper))
+ if (auto Err = launchImpl(GenericDevice, EffectiveNumThreads,
+ EffectiveNumBlocks, DynBlockMemConf.NativeSize,
+ LaunchArgs, AsyncInfoWrapper, ProfilerPtr))
return Err;
if (RecordReplay) {
@@ -1111,19 +1111,23 @@ Error GenericDeviceTy::dataDelete(void *TgtPtr, TargetAllocTy Kind,
}
Error GenericDeviceTy::dataSubmit(void *TgtPtr, const void *HstPtr,
- int64_t Size, __tgt_async_info *AsyncInfo) {
+ int64_t Size, __tgt_async_info *AsyncInfo,
+ GenericProfilerTy *ProfilerPtr) {
AsyncInfoWrapperTy AsyncInfoWrapper(*this, AsyncInfo);
- auto Err = dataSubmitImpl(TgtPtr, HstPtr, Size, AsyncInfoWrapper);
+ auto Err =
+ dataSubmitImpl(TgtPtr, HstPtr, Size, AsyncInfoWrapper, ProfilerPtr);
AsyncInfoWrapper.finalize(Err);
return Err;
}
Error GenericDeviceTy::dataRetrieve(void *HstPtr, const void *TgtPtr,
- int64_t Size, __tgt_async_info *AsyncInfo) {
+ int64_t Size, __tgt_async_info *AsyncInfo,
+ GenericProfilerTy *ProfilerPtr) {
AsyncInfoWrapperTy AsyncInfoWrapper(*this, AsyncInfo);
- auto Err = dataRetrieveImpl(HstPtr, TgtPtr, Size, AsyncInfoWrapper);
+ auto Err =
+ dataRetrieveImpl(HstPtr, TgtPtr, Size, AsyncInfoWrapper, ProfilerPtr);
AsyncInfoWrapper.finalize(Err);
return Err;
}
@@ -1142,10 +1146,12 @@ Error GenericDeviceTy::dataMemcpy(void *DstPtr, const void *SrcPtr,
Error GenericDeviceTy::dataExchange(const void *SrcPtr, GenericDeviceTy &DstDev,
void *DstPtr, int64_t Size,
- __tgt_async_info *AsyncInfo) {
+ __tgt_async_info *AsyncInfo,
+ GenericProfilerTy *ProfilerPtr) {
AsyncInfoWrapperTy AsyncInfoWrapper(*this, AsyncInfo);
- auto Err = dataExchangeImpl(SrcPtr, DstDev, DstPtr, Size, AsyncInfoWrapper);
+ auto Err = dataExchangeImpl(SrcPtr, DstDev, DstPtr, Size, AsyncInfoWrapper,
+ ProfilerPtr);
AsyncInfoWrapper.finalize(Err);
return Err;
}
diff --git a/offload/plugins-nextgen/cuda/src/rtl.cpp b/offload/plugins-nextgen/cuda/src/rtl.cpp
index 198588aa2023c..27755addebabe 100644
--- a/offload/plugins-nextgen/cuda/src/rtl.cpp
+++ b/offload/plugins-nextgen/cuda/src/rtl.cpp
@@ -140,7 +140,8 @@ struct CUDAKernelTy : public GenericKernelTy {
Error launchImpl(GenericDeviceTy &GenericDevice, uint32_t NumThreads[3],
uint32_t NumBlocks[3], uint32_t DynBlockMemSize,
KernelLaunchArgsTy &LaunchArgs,
- AsyncInfoWrapperTy &AsyncInfoWrapper) const override;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) const override;
/// Return maximum block size for maximum occupancy
Expected<uint64_t> maxGroupSize(GenericDeviceTy &,
@@ -816,7 +817,8 @@ struct CUDADeviceTy : public GenericDeviceTy {
/// Submit data to the device (host to device transfer).
Error dataSubmitImpl(void *TgtPtr, const void *HstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override {
if (auto Err = setContext())
return Err;
@@ -830,7 +832,8 @@ struct CUDADeviceTy : public GenericDeviceTy {
/// Retrieve data from the device (device to host transfer).
Error dataRetrieveImpl(void *HstPtr, const void *TgtPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override {
if (auto Err = setContext())
return Err;
@@ -860,7 +863,8 @@ struct CUDADeviceTy : public GenericDeviceTy {
/// the CUDA devices and driver allow them.
Error dataExchangeImpl(const void *SrcPtr, GenericDeviceTy &DstGenericDevice,
void *DstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override;
Error dataFillImpl(void *TgtPtr, const void *PatternPtr, int64_t PatternSize,
int64_t Size,
@@ -1394,6 +1398,9 @@ struct CUDADeviceTy : public GenericDeviceTy {
/// Returns the clock frequency for the given NVPTX device.
uint64_t getClockFrequency() const override { return 1000000000; }
+ /// Device timestamp stub for CUDA - full profiling is a future extension.
+ uint64_t getDeviceTimeStamp() override { return 0; }
+
private:
using CUDAStreamManagerTy = GenericDeviceResourceManagerTy<CUDAStreamRef>;
using CUDAEventManagerTy = GenericDeviceResourceManagerTy<CUDAEventRef>;
@@ -1488,7 +1495,8 @@ struct CUDADeviceTy : public GenericDeviceTy {
uint32_t NumBlocksAndThreads[3] = {1u, 1u, 1u};
auto Err =
CUDAKernel.launchImpl(*this, NumBlocksAndThreads, NumBlocksAndThreads,
- 0, LaunchArgs, AsyncInfoWrapper);
+ 0, LaunchArgs, AsyncInfoWrapper,
+ /*ProfilerPtr=*/nullptr);
AsyncInfoWrapper.finalize(Err);
if (Err)
@@ -1534,7 +1542,8 @@ Error CUDAKernelTy::launchImpl(GenericDeviceTy &GenericDevice,
uint32_t NumThreads[3], uint32_t NumBlocks[3],
uint32_t DynBlockMemSize,
KernelLaunchArgsTy &LaunchArgs,
- AsyncInfoWrapperTy &AsyncInfoWrapper) const {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) const {
CUDADeviceTy &CUDADevice = static_cast<CUDADeviceTy &>(GenericDevice);
CUstream Stream;
@@ -1862,7 +1871,8 @@ struct CUDAPluginTy final : public GenericPluginTy {
Error CUDADeviceTy::dataExchangeImpl(const void *SrcPtr,
GenericDeviceTy &DstGenericDevice,
void *DstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) {
if (auto Err = setContext())
return Err;
diff --git a/offload/plugins-nextgen/host/src/rtl.cpp b/offload/plugins-nextgen/host/src/rtl.cpp
index 54e8c2e6e027a..48ed9a63fe933 100644
--- a/offload/plugins-nextgen/host/src/rtl.cpp
+++ b/offload/plugins-nextgen/host/src/rtl.cpp
@@ -91,7 +91,8 @@ struct GenELF64KernelTy : public GenericKernelTy {
Error launchImpl(GenericDeviceTy &GenericDevice, uint32_t NumThreads[3],
uint32_t NumBlocks[3], uint32_t DynBlockMemSize,
KernelLaunchArgsTy &LaunchArgs,
- AsyncInfoWrapperTy &AsyncInfoWrapper) const override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) const override {
if (LaunchArgs.OmpABIVersion < OMP_KERNEL_ARG_VERSION)
return Plugin::error(ErrorCode::UNSUPPORTED,
"Incompatible kernel argument version for plugin");
@@ -271,14 +272,16 @@ struct GenELF64DeviceTy : public GenericDeviceTy {
/// Submit data to the device (host to device transfer).
Error dataSubmitImpl(void *TgtPtr, const void *HstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override {
std::memcpy(TgtPtr, HstPtr, Size);
return Plugin::success();
}
/// Retrieve data from the device (device to host transfer).
Error dataRetrieveImpl(void *HstPtr, const void *TgtPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override {
std::memcpy(HstPtr, TgtPtr, Size);
return Plugin::success();
}
@@ -293,7 +296,8 @@ struct GenELF64DeviceTy : public GenericDeviceTy {
/// supported in this plugin.
Error dataExchangeImpl(const void *SrcPtr, GenericDeviceTy &DstGenericDevice,
void *DstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override {
std::memcpy(DstPtr, SrcPtr, Size);
return Plugin::success();
}
diff --git a/offload/plugins-nextgen/level_zero/include/L0Device.h b/offload/plugins-nextgen/level_zero/include/L0Device.h
index aa102af83ff34..0a7e37c0ea47f 100644
--- a/offload/plugins-nextgen/level_zero/include/L0Device.h
+++ b/offload/plugins-nextgen/level_zero/include/L0Device.h
@@ -547,14 +547,17 @@ class L0DeviceTy final : public GenericDeviceTy {
Error queryAsyncImpl(__tgt_async_info &AsyncInfo, bool ReleaseQueue,
bool *IsQueueWorkCompleted) override;
Error dataSubmitImpl(void *TgtPtr, const void *HstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override;
Error dataRetrieveImpl(void *HstPtr, const void *TgtPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override;
Error dataMemcpyImpl(void *DstPtr, const void *SrcPtr, int64_t Size,
AsyncInfoWrapperTy &AsyncInfoWrapper) override;
Error dataExchangeImpl(const void *SrcPtr, GenericDeviceTy &DstDev,
void *DstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override;
Expected<bool>
hasPendingWorkImpl(AsyncInfoWrapperTy &AsyncInfoWrapper) override;
diff --git a/offload/plugins-nextgen/level_zero/include/L0Kernel.h b/offload/plugins-nextgen/level_zero/include/L0Kernel.h
index d9bc1bf5d727f..8d51b618043cd 100644
--- a/offload/plugins-nextgen/level_zero/include/L0Kernel.h
+++ b/offload/plugins-nextgen/level_zero/include/L0Kernel.h
@@ -83,7 +83,8 @@ class L0KernelTy : public GenericKernelTy {
Error launchImpl(GenericDeviceTy &GenericDevice, uint32_t NumThreads[3],
uint32_t NumBlocks[3], uint32_t DynBlockMemSize,
KernelLaunchArgsTy &LaunchArgs,
- AsyncInfoWrapperTy &AsyncInfoWrapper) const override;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) const override;
Error deinit() {
CALL_ZE_RET_ERROR(zeKernelDestroy, zeKernel);
return Plugin::success();
diff --git a/offload/plugins-nextgen/level_zero/src/L0Device.cpp b/offload/plugins-nextgen/level_zero/src/L0Device.cpp
index c539e7a23e523..7d4f452909b1f 100644
--- a/offload/plugins-nextgen/level_zero/src/L0Device.cpp
+++ b/offload/plugins-nextgen/level_zero/src/L0Device.cpp
@@ -347,7 +347,8 @@ Error L0DeviceTy::free(void *TgtPtr, TargetAllocTy Kind) {
}
Error L0DeviceTy::dataSubmitImpl(void *TgtPtr, const void *HstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) {
if (Size == 0)
return Plugin::success();
@@ -368,7 +369,8 @@ Error L0DeviceTy::dataSubmitImpl(void *TgtPtr, const void *HstPtr, int64_t Size,
Error L0DeviceTy::dataRetrieveImpl(void *HstPtr, const void *TgtPtr,
int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) {
if (Size == 0)
return Plugin::success();
@@ -405,7 +407,8 @@ Error L0DeviceTy::enqueueHostCallImpl(void (*Callback)(void *), void *UserData,
Error L0DeviceTy::dataExchangeImpl(const void *SrcPtr, GenericDeviceTy &DstDev,
void *DstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) {
if (auto Err =
enqueueMemCopy(DstPtr, SrcPtr, Size,
static_cast<__tgt_async_info *>(AsyncInfoWrapper)))
@@ -963,7 +966,8 @@ Error L0DeviceTy::callGlobalCtorDtorCommon(GenericPluginTy &Plugin,
uint32_t NumBlocksAndThreads[3] = {1u, 1u, 1u};
auto Err =
L0Kernel.launchImpl(*this, NumBlocksAndThreads, NumBlocksAndThreads, 0,
- LaunchArgs, AsyncInfoWrapper);
+ LaunchArgs, AsyncInfoWrapper,
+ /*ProfilerPtr=*/nullptr);
AsyncInfoWrapper.finalize(Err);
return CleanupBufferAndErr(std::move(Err));
diff --git a/offload/plugins-nextgen/level_zero/src/L0Kernel.cpp b/offload/plugins-nextgen/level_zero/src/L0Kernel.cpp
index 9c8893ecd9845..77816d635fb14 100644
--- a/offload/plugins-nextgen/level_zero/src/L0Kernel.cpp
+++ b/offload/plugins-nextgen/level_zero/src/L0Kernel.cpp
@@ -124,7 +124,8 @@ Error L0KernelTy::launchImpl(GenericDeviceTy &GenericDevice,
uint32_t NumThreads[3], uint32_t NumBlocks[3],
uint32_t DynBlockMemSize,
KernelLaunchArgsTy &LaunchArgs,
- AsyncInfoWrapperTy &AsyncInfoWrapper) const {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) const {
if (DynBlockMemSize > 0)
return Plugin::error(ErrorCode::UNSUPPORTED,
"dynamic shared memory is unsupported in L0 plugin");
More information about the llvm-branch-commits
mailing list