[llvm-branch-commits] [llvm] [Offload][AMDGPU] Wire HSA profiling into GenericProfiler abstraction (PR #226044)

via llvm-branch-commits llvm-branch-commits at lists.llvm.org
Thu Sep 24 00:18:47 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-offload

@llvm/pr-subscribers-backend-amdgpu

Author: Jan Patrick Lehr (jplehr)

<details>
<summary>Changes</summary>

Add device profiling infrastructure to the AMDGPU plugin so that the
GenericProfiler can receive nanosecond-accurate kernel execution and
data transfer timestamps from the HSA runtime.

Key changes:
- Add ProfilingInfoTy struct to transport HSA profiling data
- Add timeKernelInNsAsync/timeDataTransferInNsAsync callbacks that
  extract dispatch/copy times from HSA signals and call
  handleKernelCompletion/handleDataTransfer on the profiler
- Add getOrNullProfilerSpecificData helper to extract ProfilerData
  from AsyncInfoWrapperTy
- Add getDeviceTimeStamp() override using hsa_system_get_info
- Add getSystemTimestampInNs() for HSA system timestamp queries
- Add schedProfilerKernelTiming/schedProfilerDataTransferTiming to
  StreamSlotTy for scheduling profiler callbacks on stream slots
- Thread ProfilerSpecificData through pushKernelLaunch,
  pushMemoryCopyH2DAsync, pushMemoryCopyD2HAsync, pushMemoryCopyD2DAsync
- Extract ProfilerSpecificData in dataSubmitImpl, dataRetrieveImpl,
  dataExchangeImpl, and launchImpl
- Add stub getDeviceTimeStamp() to CUDA plugin

Assisted-by: Cursor
Assisted-by: Claude Code

---

<sub>Stack created with <a href="https://github.com/github/gh-stack">GitHub Stacks CLI</a> • <a href="https://gh.io/stacks-feedback">Give Feedback 💬</a></sub>

---

Patch is 37.73 KiB, truncated to 20.00 KiB below, full version: https://github.com/llvm/llvm-project/pull/226044.diff


11 Files Affected:

- (modified) offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp (+1) 
- (modified) offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h (+8) 
- (modified) offload/plugins-nextgen/amdgpu/src/rtl.cpp (+214-21) 
- (modified) offload/plugins-nextgen/common/include/PluginInterface.h (+14-7) 
- (modified) offload/plugins-nextgen/common/src/PluginInterface.cpp (+15-9) 
- (modified) offload/plugins-nextgen/cuda/src/rtl.cpp (+17-7) 
- (modified) offload/plugins-nextgen/host/src/rtl.cpp (+8-4) 
- (modified) offload/plugins-nextgen/level_zero/include/L0Device.h (+6-3) 
- (modified) offload/plugins-nextgen/level_zero/include/L0Kernel.h (+2-1) 
- (modified) offload/plugins-nextgen/level_zero/src/L0Device.cpp (+8-4) 
- (modified) offload/plugins-nextgen/level_zero/src/L0Kernel.cpp (+2-1) 


``````````diff
diff --git a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp
index 6cb81f06dd9c4..88273df086467 100644
--- a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp
+++ b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa.cpp
@@ -71,6 +71,7 @@ DLWRAP(hsa_amd_signal_create, 5)
 DLWRAP(hsa_amd_signal_async_handler, 5)
 DLWRAP(hsa_amd_pointer_info, 5)
 DLWRAP(hsa_amd_profiling_get_dispatch_time, 3)
+DLWRAP(hsa_amd_profiling_get_async_copy_time, 2)
 DLWRAP(hsa_amd_profiling_set_profiler_enabled, 2)
 DLWRAP(hsa_code_object_reader_create_from_memory, 3)
 DLWRAP(hsa_code_object_reader_destroy, 1)
diff --git a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h
index c736ec0759841..9a00d293bbef6 100644
--- a/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h
+++ b/offload/plugins-nextgen/amdgpu/dynamic_hsa/hsa_ext_amd.h
@@ -208,6 +208,14 @@ hsa_amd_profiling_get_dispatch_time(hsa_agent_t agent, hsa_signal_t signal,
 hsa_status_t hsa_amd_profiling_set_profiler_enabled(hsa_queue_t *queue,
                                                     int enable);
 
+typedef struct hsa_amd_profiling_async_copy_time_s {
+  uint64_t start;
+  uint64_t end;
+} hsa_amd_profiling_async_copy_time_t;
+
+hsa_status_t hsa_amd_profiling_get_async_copy_time(
+    hsa_signal_t signal, hsa_amd_profiling_async_copy_time_t *time);
+
 hsa_status_t hsa_amd_vmem_address_reserve(void **va, size_t size,
                                           uint64_t address, uint64_t flags);
 
diff --git a/offload/plugins-nextgen/amdgpu/src/rtl.cpp b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
index 888830f7dfa03..5fe8066f22836 100644
--- a/offload/plugins-nextgen/amdgpu/src/rtl.cpp
+++ b/offload/plugins-nextgen/amdgpu/src/rtl.cpp
@@ -30,6 +30,7 @@
 #include "Shared/Utils.h"
 #include "Utils/ELF.h"
 
+#include "GenericProfiler.h"
 #include "GlobalHandler.h"
 #include "OffloadAPI.h"
 #include "OpenMP/OMPT/Callback.h"
@@ -95,6 +96,89 @@ struct AMDGPUEventManagerTy;
 struct AMDGPUDeviceImageTy;
 struct AMDGPUMemoryManagerTy;
 struct AMDGPUMemoryPoolTy;
+struct AMDGPUSignalTy;
+
+/// Use to transport information to profiler timing functions.
+/// The profiler is captured as a pointer because this outlives the call that
+/// schedules the asynchronous action.
+struct ProfilingInfoTy {
+  GenericProfilerTy *Profiler;
+  hsa_agent_t Agent;
+  AMDGPUSignalTy *Signal;
+  double TicksToTime;
+  void *ProfilerSpecificData;
+};
+
+static ProfilingInfoTy *getProfilingInfo(void *Data);
+
+static std::pair<uint64_t, uint64_t>
+getKernelStartAndEndTime(const ProfilingInfoTy *Args);
+
+static std::pair<uint64_t, uint64_t>
+getCopyStartAndEndTime(const ProfilingInfoTy *Args);
+
+static Error timeKernelInNsAsync(void *Data);
+
+static Error timeDataTransferInNsAsync(void *Data) {
+  auto Args = getProfilingInfo(Data);
+  auto [Start, End] = getCopyStartAndEndTime(Args);
+  Args->Profiler->handleDataTransfer(Start, End, Args->ProfilerSpecificData);
+  return Plugin::success();
+}
+
+static void *
+getOrNullProfilerSpecificData(AsyncInfoWrapperTy &AsyncInfoWrapper) {
+  __tgt_async_info *AI = AsyncInfoWrapper;
+  return AI ? AI->ProfilerData : nullptr;
+}
+
+} // namespace plugin
+} // namespace target
+} // namespace omp
+} // namespace llvm
+
+static double setTicksToTime() {
+  uint64_t TicksFrequency = 1;
+  double TicksToTime = 1.0;
+
+  hsa_status_t Status =
+      hsa_system_get_info(HSA_SYSTEM_INFO_TIMESTAMP_FREQUENCY, &TicksFrequency);
+  if (Status == HSA_STATUS_SUCCESS)
+    TicksToTime = (double)1e9 / (double)TicksFrequency;
+
+  return TicksToTime;
+}
+
+static double TicksToTime = 1.0;
+
+static void setHSATicksToTimeConstant() { TicksToTime = setTicksToTime(); }
+
+/// Get the current HSA-based system timestamp in nanoseconds.
+/// Called by OmptTracing.cpp (from PluginOmpt) for device time queries.
+uint64_t getSystemTimestampInNs() {
+  uint64_t TimeStamp = 0;
+  hsa_status_t Status =
+      hsa_system_get_info(HSA_SYSTEM_INFO_TIMESTAMP, &TimeStamp);
+  if (Status != HSA_STATUS_SUCCESS)
+    return 0;
+  return TimeStamp;
+}
+
+/// Enable or disable HSA async copy profiling for OMPT device tracing.
+/// Called by OmptTracing.cpp when a tool activates/deactivates device tracing.
+/// Full HSA profiling integration (enabling per-copy timing signals) will be
+/// wired up in a follow-on commit.
+void setOmptAsyncCopyProfile(bool Enable) {}
+
+/// Enable or disable HSA queue kernel profiling for OMPT device tracing.
+/// Called by OmptTracing.cpp when a tool activates/deactivates device tracing.
+/// Full HSA queue profiling integration will be wired up in a follow-on commit.
+void setGlobalOmptKernelProfile(void *Device, int Enable) {}
+
+namespace llvm {
+namespace omp {
+namespace target {
+namespace plugin {
 
 namespace hsa_utils {
 
@@ -649,7 +733,8 @@ struct AMDGPUKernelTy : public GenericKernelTy {
   Error launchImpl(GenericDeviceTy &GenericDevice, uint32_t NumThreads[3],
                    uint32_t NumBlocks[3], uint32_t DynBlockMemSize,
                    KernelLaunchArgsTy &LaunchArgs,
-                   AsyncInfoWrapperTy &AsyncInfoWrapper) const override;
+                   AsyncInfoWrapperTy &AsyncInfoWrapper,
+                   GenericProfilerTy *ProfilerPtr) const override;
 
   /// Return maximum block size for maximum occupancy
   ///
@@ -1067,6 +1152,7 @@ struct AMDGPUStreamTy {
       MemcpyArgsTy MemcpyArgs;
       ReleaseBufferArgsTy ReleaseBufferArgs;
       ReleaseSignalArgsTy ReleaseSignalArgs;
+      ProfilingInfoTy ProfilerArgs;
       void *CallbackArgs;
     };
 
@@ -1104,7 +1190,31 @@ struct AMDGPUStreamTy {
       return Plugin::success();
     }
 
-    /// Register a callback to be called on compleition
+    /// Schedule kernel timing measurement via the profiler on the slot.
+    Error schedProfilerKernelTiming(GenericProfilerTy *ProfilerPtr,
+                                    hsa_agent_t Agent,
+                                    AMDGPUSignalTy *OutputSignal,
+                                    double TicksToTime,
+                                    void *ProfilerSpecificData) {
+      Callbacks.emplace_back(timeKernelInNsAsync);
+      ActionArgs.emplace_back().ProfilerArgs = ProfilingInfoTy{
+          ProfilerPtr, Agent, OutputSignal, TicksToTime, ProfilerSpecificData};
+      return Plugin::success();
+    }
+
+    /// Schedule data transfer timing via the profiler on the slot.
+    Error schedProfilerDataTransferTiming(GenericProfilerTy *ProfilerPtr,
+                                          hsa_agent_t Agent,
+                                          AMDGPUSignalTy *OutputSignal,
+                                          double TicksToTime,
+                                          void *ProfilerSpecificData) {
+      Callbacks.emplace_back(timeDataTransferInNsAsync);
+      ActionArgs.emplace_back().ProfilerArgs = ProfilingInfoTy{
+          ProfilerPtr, Agent, OutputSignal, TicksToTime, ProfilerSpecificData};
+      return Plugin::success();
+    }
+
+    /// Register a callback to be called on completion.
     Error schedCallback(AMDGPUStreamCallbackTy *Func, void *Data) {
       Callbacks.emplace_back(Func);
       ActionArgs.emplace_back().CallbackArgs = Data;
@@ -1129,6 +1239,12 @@ struct AMDGPUStreamTy {
         } else if (Callback == releaseSignalAction) {
           if (auto Err = releaseSignalAction(&ActionArg))
             return Err;
+        } else if (Callback == timeKernelInNsAsync) {
+          if (auto Err = timeKernelInNsAsync(&ActionArg))
+            return Err;
+        } else if (Callback == timeDataTransferInNsAsync) {
+          if (auto Err = timeDataTransferInNsAsync(&ActionArg))
+            return Err;
         } else if (Callback) {
           if (auto Err = Callback(ActionArg.CallbackArgs))
             return Err;
@@ -1400,7 +1516,9 @@ struct AMDGPUStreamTy {
   Error pushKernelLaunch(const AMDGPUKernelTy &Kernel, void *KernelArgs,
                          uint32_t NumThreads[3], uint32_t NumBlocks[3],
                          uint32_t GroupSize, uint64_t StackSize,
-                         AMDGPUMemoryManagerTy &MemoryManager) {
+                         AMDGPUMemoryManagerTy &MemoryManager,
+                         void *ProfilerSpecificData = nullptr,
+                         GenericProfilerTy *ProfilerPtr = nullptr) {
     if (Queue == nullptr)
       return Plugin::error(ErrorCode::INVALID_NULL_POINTER,
                            "target queue was nullptr");
@@ -1421,6 +1539,15 @@ struct AMDGPUStreamTy {
     if (auto Err = Slots[Curr].schedReleaseBuffer(KernelArgs, MemoryManager))
       return Err;
 
+#ifdef OMPT_SUPPORT
+    if (ProfilerSpecificData) {
+      if (auto Err = Slots[Curr].schedProfilerKernelTiming(
+              ProfilerPtr, Agent, OutputSignal, TicksToTime,
+              ProfilerSpecificData))
+        return Err;
+    }
+#endif
+
     // If we are running an RPC server we want to wake up the server thread
     // whenever there is a kernel running and let it sleep otherwise.
     if (Device.getRPCServer())
@@ -1483,7 +1610,9 @@ struct AMDGPUStreamTy {
   /// manager once the operation completes.
   Error pushMemoryCopyD2HAsync(void *Dst, const void *Src, void *Inter,
                                uint64_t CopySize,
-                               AMDGPUMemoryManagerTy &MemoryManager) {
+                               AMDGPUMemoryManagerTy &MemoryManager,
+                               void *ProfilerSpecificData = nullptr,
+                               GenericProfilerTy *ProfilerPtr = nullptr) {
     // Retrieve available signals for the operation's outputs.
     AMDGPUSignalTy *OutputSignals[2] = {};
     if (auto Err = SignalManager.getResources(/*Num=*/2, OutputSignals))
@@ -1546,6 +1675,8 @@ struct AMDGPUStreamTy {
   Error pushMemoryCopyH2DAsync(void *Dst, const void *Src, void *Inter,
                                uint64_t CopySize,
                                AMDGPUMemoryManagerTy &MemoryManager,
+                               void *ProfilerSpecificData = nullptr,
+                               GenericProfilerTy *ProfilerPtr = nullptr,
                                size_t NumTimes = 1) {
     // Retrieve available signals for the operation's outputs.
     AMDGPUSignalTy *OutputSignals[2] = {};
@@ -1621,7 +1752,9 @@ struct AMDGPUStreamTy {
 
   // AMDGPUDeviceTy is incomplete here, passing the underlying agent instead
   Error pushMemoryCopyD2DAsync(void *Dst, hsa_agent_t DstAgent, const void *Src,
-                               hsa_agent_t SrcAgent, uint64_t CopySize) {
+                               hsa_agent_t SrcAgent, uint64_t CopySize,
+                               void *ProfilerSpecificData = nullptr,
+                               GenericProfilerTy *ProfilerPtr = nullptr) {
     AMDGPUSignalTy *OutputSignal;
     if (auto Err = SignalManager.getResources(/*Num=*/1, &OutputSignal))
       return Err;
@@ -2288,6 +2421,8 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
     if (auto Err = initMemoryPools())
       return Err;
 
+    setHSATicksToTimeConstant();
+
     char GPUName[64];
     if (auto Err = getDeviceAttr(HSA_AGENT_INFO_NAME, GPUName))
       return Err;
@@ -2632,6 +2767,9 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
   /// Returns the clock frequency for the given AMDGPU device.
   uint64_t getClockFrequency() const override { return ClockFrequency; }
 
+  /// Returns the current HSA system timestamp for profiling.
+  uint64_t getDeviceTimeStamp() override { return getSystemTimestampInNs(); }
+
   /// Returns the HSA system timestamp frequency. Zero means unavailable.
   uint64_t getSystemTimestampFrequency() const {
     return SystemTimestampFrequency;
@@ -2843,10 +2981,13 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
 
   /// Submit data to the device (host to device transfer).
   Error dataSubmitImpl(void *TgtPtr, const void *HstPtr, int64_t Size,
-                       AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+                       AsyncInfoWrapperTy &AsyncInfoWrapper,
+                       GenericProfilerTy *ProfilerPtr) override {
     AMDGPUStreamTy *Stream = nullptr;
     void *PinnedPtr = nullptr;
 
+    auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
     // Use one-step asynchronous operation when host memory is already pinned.
     if (void *PinnedPtr =
             PinnedAllocs.getDeviceAccessiblePtrFromPinnedBuffer(HstPtr)) {
@@ -2897,15 +3038,19 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
       return Err;
 
     return Stream->pushMemoryCopyH2DAsync(TgtPtr, HstPtr, PinnedPtr, Size,
-                                          PinnedMemoryManager);
+                                          PinnedMemoryManager,
+                                          ProfilerSpecificData, ProfilerPtr);
   }
 
   /// Retrieve data from the device (device to host transfer).
   Error dataRetrieveImpl(void *HstPtr, const void *TgtPtr, int64_t Size,
-                         AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+                         AsyncInfoWrapperTy &AsyncInfoWrapper,
+                         GenericProfilerTy *ProfilerPtr) override {
     AMDGPUStreamTy *Stream = nullptr;
     void *PinnedPtr = nullptr;
 
+    auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
     // Use one-step asynchronous operation when host memory is already pinned.
     if (void *PinnedPtr =
             PinnedAllocs.getDeviceAccessiblePtrFromPinnedBuffer(HstPtr)) {
@@ -2957,7 +3102,8 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
       return Err;
 
     return Stream->pushMemoryCopyD2HAsync(HstPtr, TgtPtr, PinnedPtr, Size,
-                                          PinnedMemoryManager);
+                                          PinnedMemoryManager,
+                                          ProfilerSpecificData, ProfilerPtr);
   }
 
   Error dataMemcpyImpl(void *DstPtr, const void *SrcPtr, int64_t Size,
@@ -2990,9 +3136,12 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
   /// Exchange data between two devices within the plugin.
   Error dataExchangeImpl(const void *SrcPtr, GenericDeviceTy &DstGenericDevice,
                          void *DstPtr, int64_t Size,
-                         AsyncInfoWrapperTy &AsyncInfoWrapper) override {
+                         AsyncInfoWrapperTy &AsyncInfoWrapper,
+                         GenericProfilerTy *ProfilerPtr) override {
     AMDGPUDeviceTy &DstDevice = static_cast<AMDGPUDeviceTy &>(DstGenericDevice);
 
+    auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
     // For large transfers use synchronous behavior.
     if (Size >= OMPX_MaxAsyncCopyBytes) {
       if (AsyncInfoWrapper.hasQueue())
@@ -3021,7 +3170,8 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
       return Plugin::success();
 
     return Stream->pushMemoryCopyD2DAsync(DstPtr, DstDevice.getAgent(), SrcPtr,
-                                          getAgent(), (uint64_t)Size);
+                                          getAgent(), (uint64_t)Size,
+                                          ProfilerSpecificData, ProfilerPtr);
   }
 
   /// Insert a data fence between previous data operations and the following
@@ -3097,9 +3247,10 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
     if (auto Err = getStream(AsyncInfoWrapper, Stream))
       return Err;
 
-    return Stream->pushMemoryCopyH2DAsync(TgtPtr, PatternPtr, PinnedPtr,
-                                          PatternSize, PinnedMemoryManager,
-                                          Size / PatternSize);
+    return Stream->pushMemoryCopyH2DAsync(
+        TgtPtr, PatternPtr, PinnedPtr, PatternSize, PinnedMemoryManager,
+        /*ProfilerSpecificData=*/nullptr, /*ProfilerPtr=*/nullptr,
+        Size / PatternSize);
   }
 
   interop_spec_t selectInteropPreference(int32_t InteropType,
@@ -3592,9 +3743,9 @@ struct AMDGPUDeviceTy : public GenericDeviceTy, AMDGenericDeviceTy {
 
     KernelLaunchArgsTy LaunchArgs = {};
     uint32_t NumBlocksAndThreads[3] = {1u, 1u, 1u};
-    auto Err =
-        AMDGPUKernel.launchImpl(*this, NumBlocksAndThreads, NumBlocksAndThreads,
-                                0, LaunchArgs, AsyncInfoWrapper);
+    auto Err = AMDGPUKernel.launchImpl(
+        *this, NumBlocksAndThreads, NumBlocksAndThreads, 0, LaunchArgs,
+        AsyncInfoWrapper, /*ProfilerPtr=*/nullptr);
 
     AsyncInfoWrapper.finalize(Err);
     return Err;
@@ -4370,7 +4521,8 @@ Error AMDGPUKernelTy::launchImpl(GenericDeviceTy &GenericDevice,
                                  uint32_t NumThreads[3], uint32_t NumBlocks[3],
                                  uint32_t DynBlockMemSize,
                                  KernelLaunchArgsTy &LaunchArgs,
-                                 AsyncInfoWrapperTy &AsyncInfoWrapper) const {
+                                 AsyncInfoWrapperTy &AsyncInfoWrapper,
+                                 GenericProfilerTy *ProfilerPtr) const {
   // Cooperative kernel launch is not yet supported for AMDGPU
   if (LaunchArgs.Flags.Cooperative)
     return Plugin::error(ErrorCode::UNSUPPORTED,
@@ -4455,10 +4607,12 @@ Error AMDGPUKernelTy::launchImpl(GenericDeviceTy &GenericDevice,
   // HSA requires the group segment size to include both static and dynamic.
   uint32_t TotalBlockMemSize = getStaticBlockMemSize() + DynBlockMemSize;
 
+  auto ProfilerSpecificData = getOrNullProfilerSpecificData(AsyncInfoWrapper);
+
   // Push the kernel launch into the stream.
-  return Stream->pushKernelLaunch(*this, AllArgs, NumThreads, NumBlocks,
-                                  TotalBlockMemSize, StackSize,
-                                  ArgsMemoryManager);
+  return Stream->pushKernelLaunch(
+      *this, AllArgs, NumThreads, NumBlocks, TotalBlockMemSize, StackSize,
+      ArgsMemoryManager, ProfilerSpecificData, ProfilerPtr);
 }
 
 Error AMDGPUKernelTy::printLaunchInfoDetails(
@@ -4649,6 +4803,45 @@ void AMDGPUQueueTy::callbackError(hsa_status_t Status, hsa_queue_t *Source,
   FATAL_MESSAGE(1, "%s", toString(std::move(Err)).data());
 }
 
+/// Implementation of profiling helper functions.
+static ProfilingInfoTy *getProfilingInfo(void *Data) {
+  return reinterpret_cast<ProfilingInfoTy *>(Data);
+}
+
+static std::pair<uint64_t, uint64_t>
+getKernelStartAndEndTime(const ProfilingInfoTy *Args) {
+  hsa_amd_profiling_dispatch_time_t Time = {};
+  hsa_status_t Status = hsa_amd_profiling_get_dispatch_time(
+      Args->Agent, Args->Signal->get(), &Time);
+  if (Status != HSA_STATUS_SUCCESS)
+    return {0, 0};
+  return {static_cast<uint64_t>(Time.start * Args->TicksToTime),
+          static_cast<uint64_t>(Time.end * Args->TicksToTime)};
+}
+
+static std::pair<uint64_t, uint64_t>
+getCopyStartAndEndTime(const ProfilingInfoTy *Args) {
+  hsa_amd_profiling_async_copy_time_t Time = {};
+  hsa_status_t Status =
+      hsa_amd_profiling_get_async_copy_time(Args->Signal->get(), &Time);
+  if (Status != HSA_STATUS_SUCCESS)
+    return {0, 0};
+  return {static_cast<uint64_t>(Time.start * Args->TicksToTime),
+          static_cast<uint64_t>(Time.end * Args->TicksToTime)};
+}
+
+static Error timeKernelInNsAsync(void *Data) {
+  assert(Data && "Invalid data pointer");
+  auto ProfilerInfo = getProfilingInfo(Data);
+  assert(ProfilerInfo && "Invalid profiling info");
+  assert(ProfilerInfo->ProfilerSpecificData && "Invalid ProfilerSpecificData");
+
+  auto [StartTime, EndTime] = getKernelStartAndEndTime(ProfilerInfo);
+  ProfilerInfo->Profiler->handleKernelCompletion(
+      StartTime, EndTime, ProfilerInfo->ProfilerSpecificData);
+  return Plugin::success();
+}
+
 } // namespace plugin
 } // namespace target
 } // namespace omp
diff --git a/offload/plugins-nextgen/common/include/PluginInterface.h b/offload/plugins-nextgen/common/include/PluginInterface.h
index 17f698d0ed0a8..307198c315cb9 100644
--- a/offload/plugins-nextgen/common/include/PluginInterface.h
+++ b/offload/plugins-nextgen/common/include/PluginInterface.h
@@ -500,7 +500,8 @@ struct GenericKernelTy {
                            uint32_t NumThreads[3], uint32_t NumBlocks[3],
                            uint32_t DynBlockMemSize,
                            Kerne...
[truncated]

``````````

</details>


https://github.com/llvm/llvm-project/pull/226044


More information about the llvm-branch-commits mailing list