[llvm-branch-commits] [llvm] [Offload][AMDGPU] Wire HSA profiling into GenericProfiler abstraction (PR #226044)
via llvm-branch-commits
llvm-branch-commits at lists.llvm.org
Thu Sep 24 00:18:47 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-offload
@llvm/pr-subscribers-backend-amdgpu
Author: Jan Patrick Lehr (jplehr)
<details>
<summary>Changes</summary>
Add device profiling infrastructure to the AMDGPU plugin so that the
GenericProfiler can receive nanosecond-accurate kernel execution and
data transfer timestamps from the HSA runtime.
Key changes:
- Add ProfilingInfoTy struct to transport HSA profiling data
- Add timeKernelInNsAsync/timeDataTransferInNsAsync callbacks that
extract dispatch/copy times from HSA signals and call
handleKernelCompletion/handleDataTransfer on the profiler
- Add getOrNullProfilerSpecificData helper to extract ProfilerData
from AsyncInfoWrapperTy
- Add getDeviceTimeStamp() override using hsa_system_get_info
- Add getSystemTimestampInNs() for HSA system timestamp queries
- Add schedProfilerKernelTiming/schedProfilerDataTransferTiming to
StreamSlotTy for scheduling profiler callbacks on stream slots
- Thread ProfilerSpecificData through pushKernelLaunch,
pushMemoryCopyH2DAsync, pushMemoryCopyD2HAsync, pushMemoryCopyD2DAsync
- Extract ProfilerSpecificData in dataSubmitImpl, dataRetrieveImpl,
dataExchangeImpl, and launchImpl
- Add stub getDeviceTimeStamp() to CUDA plugin
Assisted-by: Cursor
Assisted-by: Claude Code
---
<sub>Stack created with <a href="https://github.com/github/gh-stack">GitHub Stacks CLI</a> • <a href="https://gh.io/stacks-feedback">Give Feedback 💬</a></sub>
---
Patch is 37.73 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/226044.diff
11 Files Affected:
- (modified) offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp (+1)
- (modified) offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h (+8)
- (modified) offload/plugins-nextgen/amdgpu/src/rtl.cpp (+214-21)
- (modified) offload/plugins-nextgen/common/include/PluginInterface.h (+14-7)
- (modified) offload/plugins-nextgen/common/src/PluginInterface.cpp (+15-9)
- (modified) offload/plugins-nextgen/cuda/src/rtl.cpp (+17-7)
- (modified) offload/plugins-nextgen/host/src/rtl.cpp (+8-4)
- (modified) offload/plugins-nextgen/level_zero/include/L0Device.h (+6-3)
- (modified) offload/plugins-nextgen/level_zero/include/L0Kernel.h (+2-1)
- (modified) offload/plugins-nextgen/level_zero/src/L0Device.cpp (+8-4)
- (modified) offload/plugins-nextgen/level_zero/src/L0Kernel.cpp (+2-1)
``````````diff
diff --git a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp
index 6cb81f06dd9c4..88273df086467 100644
--- a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp
+++ b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp
@@ -71,6 +71,7 @@ DLWRAP(hsa_amd_signal_create, 5)
DLWRAP(hsa_amd_signal_async_handler, 5)
DLWRAP(hsa_amd_pointer_info, 5)
DLWRAP(hsa_amd_profiling_get_dispatch_time, 3)
+DLWRAP(hsa_amd_profiling_get_async_copy_time, 2)
DLWRAP(hsa_amd_profiling_set_profiler_enabled, 2)
DLWRAP(hsa_code_object_reader_create_from_memory, 3)
DLWRAP(hsa_code_object_reader_destroy, 1)
diff --git a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h
index c736ec0759841..9a00d293bbef6 100644
--- a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h
+++ b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h
@@ -208,6 +208,14 @@ hsa_amd_profiling_get_dispatch_time(hsa_agent_t agent, hsa_signal_t signal,
hsa_status_t hsa_amd_profiling_set_profiler_enabled(hsa_queue_t *queue,
int enable);
+typedef struct hsa_amd_profiling_async_copy_time_s {
+ uint64_t start;
+ uint64_t end;
+} hsa_amd_profiling_async_copy_time_t;
+
+hsa_status_t hsa_amd_profiling_get_async_copy_time(
+ hsa_signal_t signal, hsa_amd_profiling_async_copy_time_t *time);
+
hsa_status_t hsa_amd_vmem_address_reserve(void **va, size_t size,
uint64_t address, uint64_t flags);
diff --git a/offload/plugins-nextgen/amdgpu/src/rtl.cpp b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
index 888830f7dfa03..5fe8066f22836 100644
--- a/offload/plugins-nextgen/amdgpu/src/rtl.cpp
+++ b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
@@ -30,6 +30,7 @@
#include "Shared/Utils.h"
#include "Utils/ELF.h"
+#include "GenericProfiler.h"
#include "GlobalHandler.h"
#include "OffloadAPI.h"
#include "OpenMP/OMPT/Callback.h"
@@ -95,6 +96,89 @@ struct AMDGPUEventManagerTy;
struct AMDGPUDeviceImageTy;
struct AMDGPUMemoryManagerTy;
struct AMDGPUMemoryPoolTy;
+struct AMDGPUSignalTy;
+
+/// Use to transport information to profiler timing functions.
+/// The profiler is captured as a pointer because this outlives the call that
+/// schedules the asynchronous action.
+struct ProfilingInfoTy {
+ GenericProfilerTy *Profiler;
+ hsa_agent_t Agent;
+ AMDGPUSignalTy *Signal;
+ double TicksToTime;
+ void *ProfilerSpecificData;
+};
+
+static ProfilingInfoTy *getProfilingInfo(void *Data);
+
+static std::pair<uint64_t, uint64_t>
+getKernelStartAndEndTime(const ProfilingInfoTy *Args);
+
+static std::pair<uint64_t, uint64_t>
+getCopyStartAndEndTime(const ProfilingInfoTy *Args);
+
+static Error timeKernelInNsAsync(void *Data);
+
+static Error timeDataTransferInNsAsync(void *Data) {
+ auto Args = getProfilingInfo(Data);
+ auto [Start, End] = getCopyStartAndEndTime(Args);
+ Args->Profiler->handleDataTransfer(Start, End, Args->ProfilerSpecificData);
+ return Plugin::success();
+}
+
+static void *
+getOrNullProfilerSpecificData(AsyncInfoWrapperTy &AsyncInfoWrapper) {
+ __tgt_async_info *AI = AsyncInfoWrapper;
+ return AI ? AI->ProfilerData : nullptr;
+}
+
+} // namespace plugin
+} // namespace target
+} // namespace omp
+} // namespace llvm
+
+static double setTicksToTime() {
+ uint64_t TicksFrequency = 1;
+ double TicksToTime = 1.0;
+
+ hsa_status_t Status =
+ hsa_system_get_info(HSA_SYSTEM_INFO_TIMESTAMP_FREQUENCY, &TicksFrequency);
+ if (Status == HSA_STATUS_SUCCESS)
+ TicksToTime = (double)1e9 / (double)TicksFrequency;
+
+ return TicksToTime;
+}
+
+static double TicksToTime = 1.0;
+
+static void setHSATicksToTimeConstant() { TicksToTime = setTicksToTime(); }
+
+/// Get the current HSA-based system timestamp in nanoseconds.
+/// Called by OmptTracing.cpp (from PluginOmpt) for device time queries.
+uint64_t getSystemTimestampInNs() {
+ uint64_t TimeStamp = 0;
+ hsa_status_t Status =
+ hsa_system_get_info(HSA_SYSTEM_INFO_TIMESTAMP, &TimeStamp);
+ if (Status != HSA_STATUS_SUCCESS)
+ return 0;
+ return TimeStamp;
+}
+
+/// Enable or disable HSA async copy profiling for OMPT device tracing.
+/// Called by OmptTracing.cpp when a tool activates/deactivates device tracing.
+/// Full HSA profiling integration (enabling per-copy timing signals) will be
+/// wired up in a follow-on commit.
+void setOmptAsyncCopyProfile(bool Enable) {}
+
+/// Enable or disable HSA queue kernel profiling for OMPT device tracing.
+/// Called by OmptTracing.cpp when a tool activates/deactivates device tracing.
+/// Full HSA queue profiling integration will be wired up in a follow-on commit.
+void setGlobalOmptKernelProfile(void *Device, int Enable) {}
+
+namespace llvm {
+namespace omp {
+namespace target {
+namespace plugin {
namespace hsa_utils {
@@ -649,7 +733,8 @@ struct AMDGPUKernelTy : public GenericKernelTy {
Error launchImpl(GenericDeviceTy &GenericDevice, uint32_t NumThreads[3],
uint32_t NumBlocks[3], uint32_t DynBlockMemSize,
KernelLaunchArgsTy &LaunchArgs,
- AsyncInfoWrapperTy &AsyncInfoWrapper) const override;
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) const override;
/// Return maximum block size for maximum occupancy
///
@@ -1067,6 +1152,7 @@ struct AMDGPUStreamTy {
MemcpyArgsTy MemcpyArgs;
ReleaseBufferArgsTy ReleaseBufferArgs;
ReleaseSignalArgsTy ReleaseSignalArgs;
+ ProfilingInfoTy ProfilerArgs;
void *CallbackArgs;
};
@@ -1104,7 +1190,31 @@ struct AMDGPUStreamTy {
return Plugin::success();
}
- /// Register a callback to be called on compleition
+ /// Schedule kernel timing measurement via the profiler on the slot.
+ Error schedProfilerKernelTiming(GenericProfilerTy *ProfilerPtr,
+ hsa_agent_t Agent,
+ AMDGPUSignalTy *OutputSignal,
+ double TicksToTime,
+ void *ProfilerSpecificData) {
+ Callbacks.emplace_back(timeKernelInNsAsync);
+ ActionArgs.emplace_back().ProfilerArgs = ProfilingInfoTy{
+ ProfilerPtr, Agent, OutputSignal, TicksToTime, ProfilerSpecificData};
+ return Plugin::success();
+ }
+
+ /// Schedule data transfer timing via the profiler on the slot.
+ Error schedProfilerDataTransferTiming(GenericProfilerTy *ProfilerPtr,
+ hsa_agent_t Agent,
+ AMDGPUSignalTy *OutputSignal,
+ double TicksToTime,
+ void *ProfilerSpecificData) {
+ Callbacks.emplace_back(timeDataTransferInNsAsync);
+ ActionArgs.emplace_back().ProfilerArgs = ProfilingInfoTy{
+ ProfilerPtr, Agent, OutputSignal, TicksToTime, ProfilerSpecificData};
+ return Plugin::success();
+ }
+
+ /// Register a callback to be called on completion.
Error schedCallback(AMDGPUStreamCallbackTy *Func, void *Data) {
Callbacks.emplace_back(Func);
ActionArgs.emplace_back().CallbackArgs = Data;
@@ -1129,6 +1239,12 @@ struct AMDGPUStreamTy {
} else if (Callback == releaseSignalAction) {
if (auto Err = releaseSignalAction(&ActionArg))
return Err;
+ } else if (Callback == timeKernelInNsAsync) {
+ if (auto Err = timeKernelInNsAsync(&ActionArg))
+ return Err;
+ } else if (Callback == timeDataTransferInNsAsync) {
+ if (auto Err = timeDataTransferInNsAsync(&ActionArg))
+ return Err;
} else if (Callback) {
if (auto Err = Callback(ActionArg.CallbackArgs))
return Err;
@@ -1400,7 +1516,9 @@ struct AMDGPUStreamTy {
Error pushKernelLaunch(const AMDGPUKernelTy &Kernel, void *KernelArgs,
uint32_t NumThreads[3], uint32_t NumBlocks[3],
uint32_t GroupSize, uint64_t StackSize,
- AMDGPUMemoryManagerTy &MemoryManager) {
+ AMDGPUMemoryManagerTy &MemoryManager,
+ void *ProfilerSpecificData = nullptr,
+ GenericProfilerTy *ProfilerPtr = nullptr) {
if (Queue == nullptr)
return Plugin::error(ErrorCode::INVALID_NULL_POINTER,
"target queue was nullptr");
@@ -1421,6 +1539,15 @@ struct AMDGPUStreamTy {
if (auto Err = Slots[Curr].schedReleaseBuffer(KernelArgs, MemoryManager))
return Err;
+#ifdef OMPT_SUPPORT
+ if (ProfilerSpecificData) {
+ if (auto Err = Slots[Curr].schedProfilerKernelTiming(
+ ProfilerPtr, Agent, OutputSignal, TicksToTime,
+ ProfilerSpecificData))
+ return Err;
+ }
+#endif
+
// If we are running an RPC server we want to wake up the server thread
// whenever there is a kernel running and let it sleep otherwise.
if (Device.getRPCServer())
@@ -1483,7 +1610,9 @@ struct AMDGPUStreamTy {
/// manager once the operation completes.
Error pushMemoryCopyD2HAsync(void *Dst, const void *Src, void *Inter,
uint64_t CopySize,
- AMDGPUMemoryManagerTy &MemoryManager) {
+ AMDGPUMemoryManagerTy &MemoryManager,
+ void *ProfilerSpecificData = nullptr,
+ GenericProfilerTy *ProfilerPtr = nullptr) {
// Retrieve available signals for the operation's outputs.
AMDGPUSignalTy *OutputSignals[2] = {};
if (auto Err = SignalManager.getResources(/*Num=*/2, OutputSignals))
@@ -1546,6 +1675,8 @@ struct AMDGPUStreamTy {
Error pushMemoryCopyH2DAsync(void *Dst, const void *Src, void *Inter,
uint64_t CopySize,
AMDGPUMemoryManagerTy &MemoryManager,
+ void *ProfilerSpecificData = nullptr,
+ GenericProfilerTy *ProfilerPtr = nullptr,
size_t NumTimes = 1) {
// Retrieve available signals for the operation's outputs.
AMDGPUSignalTy *OutputSignals[2] = {};
@@ -1621,7 +1752,9 @@ struct AMDGPUStreamTy {
// AMDGPUDeviceTy is incomplete here, passing the underlying agent instead
Error pushMemoryCopyD2DAsync(void *Dst, hsa_agent_t DstAgent, const void *Src,
- hsa_agent_t SrcAgent, uint64_t CopySize) {
+ hsa_agent_t SrcAgent, uint64_t CopySize,
+ void *ProfilerSpecificData = nullptr,
+ GenericProfilerTy *ProfilerPtr = nullptr) {
AMDGPUSignalTy *OutputSignal;
if (auto Err = SignalManager.getResources(/*Num=*/1, &OutputSignal))
return Err;
@@ -2288,6 +2421,8 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
if (auto Err = initMemoryPools())
return Err;
+ setHSATicksToTimeConstant();
+
char GPUName[64];
if (auto Err = getDeviceAttr(HSA_AGENT_INFO_NAME, GPUName))
return Err;
@@ -2632,6 +2767,9 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
/// Returns the clock frequency for the given AMDGPU device.
uint64_t getClockFrequency() const override { return ClockFrequency; }
+ /// Returns the current HSA system timestamp for profiling.
+ uint64_t getDeviceTimeStamp() override { return getSystemTimestampInNs(); }
+
/// Returns the HSA system timestamp frequency. Zero means unavailable.
uint64_t getSystemTimestampFrequency() const {
return SystemTimestampFrequency;
@@ -2843,10 +2981,13 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
/// Submit data to the device (host to device transfer).
Error dataSubmitImpl(void *TgtPtr, const void *HstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override {
AMDGPUStreamTy *Stream = nullptr;
void *PinnedPtr = nullptr;
+ auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
// Use one-step asynchronous operation when host memory is already pinned.
if (void *PinnedPtr =
PinnedAllocs.getDeviceAccessiblePtrFromPinnedBuffer(HstPtr)) {
@@ -2897,15 +3038,19 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
return Err;
return Stream->pushMemoryCopyH2DAsync(TgtPtr, HstPtr, PinnedPtr, Size,
- PinnedMemoryManager);
+ PinnedMemoryManager,
+ ProfilerSpecificData, ProfilerPtr);
}
/// Retrieve data from the device (device to host transfer).
Error dataRetrieveImpl(void *HstPtr, const void *TgtPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override {
AMDGPUStreamTy *Stream = nullptr;
void *PinnedPtr = nullptr;
+ auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
// Use one-step asynchronous operation when host memory is already pinned.
if (void *PinnedPtr =
PinnedAllocs.getDeviceAccessiblePtrFromPinnedBuffer(HstPtr)) {
@@ -2957,7 +3102,8 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
return Err;
return Stream->pushMemoryCopyD2HAsync(HstPtr, TgtPtr, PinnedPtr, Size,
- PinnedMemoryManager);
+ PinnedMemoryManager,
+ ProfilerSpecificData, ProfilerPtr);
}
Error dataMemcpyImpl(void *DstPtr, const void *SrcPtr, int64_t Size,
@@ -2990,9 +3136,12 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
/// Exchange data between two devices within the plugin.
Error dataExchangeImpl(const void *SrcPtr, GenericDeviceTy &DstGenericDevice,
void *DstPtr, int64_t Size,
- AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) override {
AMDGPUDeviceTy &DstDevice = static_cast<AMDGPUDeviceTy &>(DstGenericDevice);
+ auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
// For large transfers use synchronous behavior.
if (Size >= OMPX_MaxAsyncCopyBytes) {
if (AsyncInfoWrapper.hasQueue())
@@ -3021,7 +3170,8 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
return Plugin::success();
return Stream->pushMemoryCopyD2DAsync(DstPtr, DstDevice.getAgent(), SrcPtr,
- getAgent(), (uint64_t)Size);
+ getAgent(), (uint64_t)Size,
+ ProfilerSpecificData, ProfilerPtr);
}
/// Insert a data fence between previous data operations and the following
@@ -3097,9 +3247,10 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
if (auto Err = getStream(AsyncInfoWrapper, Stream))
return Err;
- return Stream->pushMemoryCopyH2DAsync(TgtPtr, PatternPtr, PinnedPtr,
- PatternSize, PinnedMemoryManager,
- Size / PatternSize);
+ return Stream->pushMemoryCopyH2DAsync(
+ TgtPtr, PatternPtr, PinnedPtr, PatternSize, PinnedMemoryManager,
+ /*ProfilerSpecificData=*/nullptr, /*ProfilerPtr=*/nullptr,
+ Size / PatternSize);
}
interop_spec_t selectInteropPreference(int32_t InteropType,
@@ -3592,9 +3743,9 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
KernelLaunchArgsTy LaunchArgs = {};
uint32_t NumBlocksAndThreads[3] = {1u, 1u, 1u};
- auto Err =
- AMDGPUKernel.launchImpl(*this, NumBlocksAndThreads, NumBlocksAndThreads,
- 0, LaunchArgs, AsyncInfoWrapper);
+ auto Err = AMDGPUKernel.launchImpl(
+ *this, NumBlocksAndThreads, NumBlocksAndThreads, 0, LaunchArgs,
+ AsyncInfoWrapper, /*ProfilerPtr=*/nullptr);
AsyncInfoWrapper.finalize(Err);
return Err;
@@ -4370,7 +4521,8 @@ Error AMDGPUKernelTy::launchImpl(GenericDeviceTy &GenericDevice,
uint32_t NumThreads[3], uint32_t NumBlocks[3],
uint32_t DynBlockMemSize,
KernelLaunchArgsTy &LaunchArgs,
- AsyncInfoWrapperTy &AsyncInfoWrapper) const {
+ AsyncInfoWrapperTy &AsyncInfoWrapper,
+ GenericProfilerTy *ProfilerPtr) const {
// Cooperative kernel launch is not yet supported for AMDGPU
if (LaunchArgs.Flags.Cooperative)
return Plugin::error(ErrorCode::UNSUPPORTED,
@@ -4455,10 +4607,12 @@ Error AMDGPUKernelTy::launchImpl(GenericDeviceTy &GenericDevice,
// HSA requires the group segment size to include both static and dynamic.
uint32_t TotalBlockMemSize = getStaticBlockMemSize() + DynBlockMemSize;
+ auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
// Push the kernel launch into the stream.
- return Stream->pushKernelLaunch(*this, AllArgs, NumThreads, NumBlocks,
- TotalBlockMemSize, StackSize,
- ArgsMemoryManager);
+ return Stream->pushKernelLaunch(
+ *this, AllArgs, NumThreads, NumBlocks, TotalBlockMemSize, StackSize,
+ ArgsMemoryManager, ProfilerSpecificData, ProfilerPtr);
}
Error AMDGPUKernelTy::printLaunchInfoDetails(
@@ -4649,6 +4803,45 @@ void AMDGPUQueueTy::callbackError(hsa_status_t Status, hsa_queue_t *Source,
FATAL_MESSAGE(1, "%s", toString(std::move(Err)).data());
}
+/// Implementation of profiling helper functions.
+static ProfilingInfoTy *getProfilingInfo(void *Data) {
+ return reinterpret_cast<ProfilingInfoTy *>(Data);
+}
+
+static std::pair<uint64_t, uint64_t>
+getKernelStartAndEndTime(const ProfilingInfoTy *Args) {
+ hsa_amd_profiling_dispatch_time_t Time = {};
+ hsa_status_t Status = hsa_amd_profiling_get_dispatch_time(
+ Args->Agent, Args->Signal->get(), &Time);
+ if (Status != HSA_STATUS_SUCCESS)
+ return {0, 0};
+ return {static_cast<uint64_t>(Time.start * Args->TicksToTime),
+ static_cast<uint64_t>(Time.end * Args->TicksToTime)};
+}
+
+static std::pair<uint64_t, uint64_t>
+getCopyStartAndEndTime(const ProfilingInfoTy *Args) {
+ hsa_amd_profiling_async_copy_time_t Time = {};
+ hsa_status_t Status =
+ hsa_amd_profiling_get_async_copy_time(Args->Signal->get(), &Time);
+ if (Status != HSA_STATUS_SUCCESS)
+ return {0, 0};
+ return {static_cast<uint64_t>(Time.start * Args->TicksToTime),
+ static_cast<uint64_t>(Time.end * Args->TicksToTime)};
+}
+
+static Error timeKernelInNsAsync(void *Data) {
+ assert(Data && "Invalid data pointer");
+ auto ProfilerInfo = getProfilingInfo(Data);
+ assert(ProfilerInfo && "Invalid profiling info");
+ assert(ProfilerInfo->ProfilerSpecificData && "Invalid ProfilerSpecificData");
+
+ auto [StartTime, EndTime] = getKernelStartAndEndTime(ProfilerInfo);
+ ProfilerInfo->Profiler->handleKernelCompletion(
+ StartTime, EndTime, ProfilerInfo->ProfilerSpecificData);
+ return Plugin::success();
+}
+
} // namespace plugin
} // namespace target
} // namespace omp
diff --git a/offload/plugins-nextgen/common/include/PluginInterface.h b/offload/plugins-nextgen/common/include/PluginInterface.h
index 17f698d0ed0a8..307198c315cb9 100644
--- a/offload/plugins-nextgen/common/include/PluginInterface.h
+++ b/offload/plugins-nextgen/common/include/PluginInterface.h
@@ -500,7 +500,8 @@ struct GenericKernelTy {
uint32_t NumThreads[3], uint32_t NumBlocks[3],
uint32_t DynBlockMemSize,
Kerne...
[truncated]
``````````
</details>
https://github.com/llvm/llvm-project/pull/226044
More information about the llvm-branch-commits
mailing list