[llvm] [offload][omp] Move reading _kernel_environment to libomptarget (PR #222606)

Alex Duran via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 21 06:55:57 PDT 2026


https://github.com/adurang updated https://github.com/llvm/llvm-project/pull/222606

>From c358556b32b570e02158f13f2d76d117bd283bd8 Mon Sep 17 00:00:00 2001
From: "Duran, Alex" <alejandro.duran at intel.com>
Date: Thu, 10 Sep 2026 02:54:39 -0700
Subject: [PATCH 1/3] [offload][omp] Move reading _kernel_environment to
 libomptarget

---
 offload/include/PluginManager.h               |   6 +
 offload/include/device.h                      |  19 +++
 offload/libompaccsupport/PluginManager.cpp    |  66 ++++++++++
 offload/libompaccsupport/device.cpp           |   1 +
 .../common/include/PluginInterface.h          | 121 +++++++++---------
 .../common/src/PluginInterface.cpp            |  73 ++++-------
 .../common/src/RecordReplay.cpp               |   2 +-
 offload/plugins-nextgen/cuda/src/rtl.cpp      |  11 +-
 offload/plugins-nextgen/host/src/rtl.cpp      |   6 -
 9 files changed, 185 insertions(+), 120 deletions(-)

diff --git a/offload/include/PluginManager.h b/offload/include/PluginManager.h
index 6c6fdebe76dff..7e15c68d4a916 100644
--- a/offload/include/PluginManager.h
+++ b/offload/include/PluginManager.h
@@ -150,6 +150,9 @@ struct PluginManager {
     return count;
   }
 
+  /// Return the host (GenELF64) plugin, or nullptr if it wasn't built.
+  GenericPluginTy *getHostPlugin() const { return HostPlugin; }
+
 private:
   bool RTLsLoaded = false;
   llvm::SmallVector<__tgt_bin_desc *> DelayedBinDesc;
@@ -157,6 +160,9 @@ struct PluginManager {
   // List of all plugins, in use or not.
   llvm::SmallVector<std::unique_ptr<GenericPluginTy>> Plugins;
 
+  // The host (GenELF64) plugin.
+  GenericPluginTy *HostPlugin = nullptr;
+
   // Mapping of plugins to the OpenMP device identifier.
   llvm::DenseMap<std::pair<const GenericPluginTy *, int32_t>, int32_t>
       DeviceIds;
diff --git a/offload/include/device.h b/offload/include/device.h
index 266a2a675df0c..3d76b742a59b9 100644
--- a/offload/include/device.h
+++ b/offload/include/device.h
@@ -39,6 +39,7 @@
 using GenericPluginTy = llvm::omp::target::plugin::GenericPluginTy;
 using DeviceInfo = llvm::omp::target::plugin::DeviceInfo;
 using InfoTreeNode = llvm::omp::target::plugin::InfoTreeNode;
+using KernelLaunchInfoTy = llvm::omp::target::plugin::KernelLaunchInfoTy;
 
 // Forward declarations.
 struct __tgt_bin_desc;
@@ -184,6 +185,20 @@ struct DeviceTy {
     return std::get<T>(Entry->Value);
   }
 
+  /// Record the launch-geometry properties for the kernel at \p KernelPtr,
+  /// read once at registration time from its "<name>_kernel_environment"
+  /// device global.
+  void setKernelLaunchInfo(void *KernelPtr, KernelLaunchInfoTy Info) {
+    (*KernelLaunchInfoMap.getExclusiveAccessor())[KernelPtr] = Info;
+  }
+
+  /// Return the launch-geometry properties recorded for the kernel at
+  /// \p KernelPtr, or a default-constructed KernelLaunchInfoTy if none were
+  /// recorded.
+  KernelLaunchInfoTy getKernelLaunchInfo(void *KernelPtr) {
+    return (*KernelLaunchInfoMap.getExclusiveAccessor())[KernelPtr];
+  }
+
 private:
   /// Deinitialize the device (and plugin).
   void deinit();
@@ -193,6 +208,10 @@ struct DeviceTy {
       llvm::DenseMap<llvm::StringRef, OffloadEntryTy>;
   ProtectedObj<DeviceOffloadEntriesMapTy> DeviceOffloadEntries;
 
+  /// Launch-geometry properties for each kernel registered on this device.
+  using KernelLaunchInfoMapTy = llvm::DenseMap<void *, KernelLaunchInfoTy>;
+  ProtectedObj<KernelLaunchInfoMapTy> KernelLaunchInfoMap;
+
   /// Handler to collect and organize host-2-device mapping information.
   MappingInfoTy MappingInfo;
 
diff --git a/offload/libompaccsupport/PluginManager.cpp b/offload/libompaccsupport/PluginManager.cpp
index 41b653a60adfd..5c495b666d409 100644
--- a/offload/libompaccsupport/PluginManager.cpp
+++ b/offload/libompaccsupport/PluginManager.cpp
@@ -13,12 +13,15 @@
 #include "PluginManager.h"
 #include "OffloadPolicy.h"
 #include "Shared/Debug.h"
+#include "Shared/Environment.h"
 #include "Shared/Profile.h"
 #include "device.h"
 
 #include "llvm/Support/Error.h"
 #include "llvm/Support/ErrorHandling.h"
+#include <algorithm>
 #include <memory>
+#include <string>
 
 using namespace llvm;
 using namespace llvm::sys;
@@ -44,6 +47,8 @@ void PluginManager::init() {
   do {                                                                         \
     Plugins.emplace_back(                                                      \
         std::unique_ptr<GenericPluginTy>(createPlugin_##Name()));              \
+    if (strcmp(#Name, "host") == 0)                                            \
+      HostPlugin = Plugins.back().get();                                       \
   } while (false);
 #include "Shared/Targets.def"
 
@@ -461,6 +466,67 @@ static int loadImagesOntoDevice(DeviceTy &Device) {
           if (Device.RTL->get_function(Binary, Entry.SymbolName,
                                        &DeviceEntry.Address) != OFFLOAD_SUCCESS)
             REPORT() << "Failed to load kernel " << Entry.SymbolName;
+
+          // Read this kernel's launch-geometry properties once, from its
+          // "<name>_kernel_environment" device global, and cache them on
+          // the device for use at launch time.
+          std::string EnvName =
+              std::string(Entry.SymbolName) + "_kernel_environment";
+          KernelEnvironmentTy KernelEnv{};
+          void *KernelEnvPtr = nullptr;
+          bool ReadOk =
+              Device.RTL->get_global(Binary, sizeof(KernelEnv), EnvName.c_str(),
+                                     &KernelEnvPtr) == OFFLOAD_SUCCESS &&
+              Device.RTL->data_retrieve(DeviceId, &KernelEnv, KernelEnvPtr,
+                                        sizeof(KernelEnv)) == OFFLOAD_SUCCESS;
+          if (!ReadOk) {
+            KernelEnv = KernelEnvironmentTy{};
+            // If no kernel environment is found, the host kernels are expected
+            // to run in Generic execution mode. Other backends expect to be run
+            // in Bare mode.
+            if (Device.RTL == PM->getHostPlugin())
+              KernelEnv.Configuration.ExecMode =
+                  llvm::omp::OMP_TGT_EXEC_MODE_GENERIC;
+            else
+              KernelEnv.Configuration.ExecMode =
+                  llvm::omp::OMP_TGT_EXEC_MODE_BARE;
+            ODBG(ODT_Mapping)
+                << "Failed to read kernel environment for '" << Entry.SymbolName
+                << "', using default "
+                << KernelLaunchInfoTy::getExecutionModeName(
+                       static_cast<llvm::omp::OMPTgtExecModeFlags>(
+                           KernelEnv.Configuration.ExecMode))
+                << " execution mode";
+          }
+
+          llvm::omp::target::plugin::GenericDeviceTy &GenericDevice =
+              Device.RTL->getDevice(DeviceId);
+          auto *Kernel =
+              reinterpret_cast<llvm::omp::target::plugin::GenericKernelTy *>(
+                  DeviceEntry.Address);
+          const auto &Cfg = KernelEnv.Configuration;
+          KernelLaunchInfoTy LaunchInfo;
+          LaunchInfo.Mode =
+              static_cast<llvm::omp::OMPTgtExecModeFlags>(Cfg.ExecMode);
+          LaunchInfo.ReductionDataSize = Cfg.ReductionDataSize;
+          // Max = Config.Max > 0 ? min(Config.Max, Device.Max) : Device.Max,
+          // further clamped to the kernel function's own driver-reported
+          // maximum.
+          LaunchInfo.MaxNumThreads =
+              std::min(Cfg.MaxThreads > 0
+                           ? std::min(Cfg.MaxThreads,
+                                      int32_t(GenericDevice.getThreadLimit()))
+                           : GenericDevice.getThreadLimit(),
+                       Kernel->getMaxThreads());
+          // Pref = Config.Pref > 0 ? max(Config.Pref, Device.Pref)
+          //                        : Device.Pref.
+          LaunchInfo.PreferredNumThreads =
+              Cfg.MinThreads > 0
+                  ? std::max(Cfg.MinThreads,
+                             int32_t(GenericDevice.getDefaultNumThreads()))
+                  : GenericDevice.getDefaultNumThreads();
+
+          Device.setKernelLaunchInfo(DeviceEntry.Address, LaunchInfo);
         }
         ODBG(ODT_Mapping) << "Entry point " << Entry.Address << " maps to"
                           << (Entry.Size ? " global" : "") << " "
diff --git a/offload/libompaccsupport/device.cpp b/offload/libompaccsupport/device.cpp
index 688746477861c..ff51d346436a9 100644
--- a/offload/libompaccsupport/device.cpp
+++ b/offload/libompaccsupport/device.cpp
@@ -407,6 +407,7 @@ int32_t DeviceTy::launchKernel(void *TgtEntryPtr, void **TgtVarsPtr,
   LaunchArgs.Flags.StrictBlocks = KernelArgs.Flags.StrictBlocks;
   LaunchArgs.Flags.StrictThreads = KernelArgs.Flags.StrictThreads;
   LaunchArgs.Flags.DynCGroupMemFallback = KernelArgs.Flags.DynCGroupMemFallback;
+  LaunchArgs.KernelEnvironment = getKernelLaunchInfo(TgtEntryPtr);
 
   if (KernelArgs.Flags.IsCUDA) {
     // Kernel languages (CUDA/HIP) pass an already-flattened argument-pointer
diff --git a/offload/plugins-nextgen/common/include/PluginInterface.h b/offload/plugins-nextgen/common/include/PluginInterface.h
index 0881fa3fd7627..cd9df63e87d82 100644
--- a/offload/plugins-nextgen/common/include/PluginInterface.h
+++ b/offload/plugins-nextgen/common/include/PluginInterface.h
@@ -14,6 +14,7 @@
 #include <cstddef>
 #include <cstdint>
 #include <deque>
+#include <limits>
 #include <list>
 #include <map>
 #include <shared_mutex>
@@ -428,6 +429,46 @@ class DeviceImageTy {
   }
 };
 
+struct KernelLaunchInfoTy {
+  uint32_t MaxNumThreads = 0;
+  uint32_t PreferredNumThreads = 0;
+  uint32_t ReductionDataSize = 0;
+  /// Defaults to OMP_TGT_EXEC_MODE_BARE.
+  OMPTgtExecModeFlags Mode = OMP_TGT_EXEC_MODE_BARE;
+
+  /// Indicate if the kernel works in Bare, Generic SPMD, Generic, No-Loop
+  /// or SPMD mode.
+  bool isBareMode() const { return Mode == OMP_TGT_EXEC_MODE_BARE; }
+  bool isGenericMode() const { return Mode == OMP_TGT_EXEC_MODE_GENERIC; }
+  bool isGenericSPMDMode() const {
+    return Mode == OMP_TGT_EXEC_MODE_GENERIC_SPMD;
+  }
+  bool isSPMDMode() const { return Mode == OMP_TGT_EXEC_MODE_SPMD; }
+  bool isNoLoopMode() const { return Mode == OMP_TGT_EXEC_MODE_SPMD_NO_LOOP; }
+
+  static const char *getExecutionModeName(OMPTgtExecModeFlags Mode) {
+    switch (Mode) {
+    case OMP_TGT_EXEC_MODE_BARE:
+      return "BARE";
+    case OMP_TGT_EXEC_MODE_SPMD:
+      return "SPMD";
+    case OMP_TGT_EXEC_MODE_GENERIC:
+      return "Generic";
+    case OMP_TGT_EXEC_MODE_GENERIC_SPMD:
+      return "Generic-SPMD";
+    case OMP_TGT_EXEC_MODE_SPMD_NO_LOOP:
+      return "SPMD-No-Loop";
+    }
+    return "Unknown";
+  }
+
+  /// Return the display name of this kernel's execution mode, for
+  /// debug/info logging only.
+  const char *getExecutionModeName() const {
+    return getExecutionModeName(Mode);
+  }
+};
+
 /// The subset of KernelArgsTy fields the plugin interface needs to launch a
 /// kernel, plus the resolved argument-pointer array. Unlike KernelArgsTy,
 /// this struct is populated by libomptarget on the stack for every launch,
@@ -458,6 +499,7 @@ struct KernelLaunchArgsTy {
   uint32_t UserNumBlocks[3] = {0, 0, 0};
   /// User-requested number of threads (for x,y,z dimension).
   uint32_t UserThreadLimit[3] = {0, 0, 0};
+  KernelLaunchInfoTy KernelEnvironment;
   struct {
     uint64_t Cooperative : 1; // Was this kernel spawned as cooperative.
     uint64_t StrictBlocks : 1; // The user-requested number of blocks is strict.
@@ -475,9 +517,8 @@ struct KernelLaunchArgsTy {
 /// should define the specific kernel class, derive from this generic one, and
 /// implement the necessary virtual function members.
 struct GenericKernelTy {
-  /// Construct a kernel with a name and a execution mode.
-  GenericKernelTy(StringRef Name)
-      : Name(Name), PreferredNumThreads(0), MaxNumThreads(0) {}
+  /// Construct a kernel with a name.
+  GenericKernelTy(StringRef Name) : Name(Name) {}
 
   virtual ~GenericKernelTy() {}
 
@@ -518,8 +559,11 @@ struct GenericKernelTy {
   /// Get the size of the static per-block memory consumed by the kernel.
   uint32_t getStaticBlockMemSize() const { return StaticBlockMemSize; };
 
-  /// Get the maximum number of threads per block that this kernel may use.
-  uint32_t getMaxThreads() const { return MaxNumThreads; }
+  /// Return the maximum number of threads per block that this kernel's
+  /// underlying device function may run, as reported by the driver/backend.
+  virtual uint32_t getMaxThreads() const {
+    return std::numeric_limits<uint32_t>::max();
+  }
 
   /// Get the kernel image.
   DeviceImageTy &getImage() const {
@@ -527,11 +571,6 @@ struct GenericKernelTy {
     return *ImagePtr;
   }
 
-  /// Return the kernel environment object for kernel \p Name.
-  const KernelEnvironmentTy &getKernelEnvironmentForKernel() {
-    return KernelEnvironment;
-  }
-
   /// Return a device pointer to a new kernel launch environment.
   ///
   /// \p NumBlocks0 is the number of blocks for this launch and is used to size
@@ -555,23 +594,6 @@ struct GenericKernelTy {
   }
 
 protected:
-  /// Get the execution mode name of the kernel.
-  const char *getExecutionModeName() const {
-    switch (KernelEnvironment.Configuration.ExecMode) {
-    case OMP_TGT_EXEC_MODE_BARE:
-      return "BARE";
-    case OMP_TGT_EXEC_MODE_SPMD:
-      return "SPMD";
-    case OMP_TGT_EXEC_MODE_GENERIC:
-      return "Generic";
-    case OMP_TGT_EXEC_MODE_GENERIC_SPMD:
-      return "Generic-SPMD";
-    case OMP_TGT_EXEC_MODE_SPMD_NO_LOOP:
-      return "SPMD-No-Loop";
-    }
-    llvm_unreachable("Unknown execution mode!");
-  }
-
   /// Prints generic kernel launch information.
   Error printLaunchInfo(GenericDeviceTy &GenericDevice,
                         const KernelLaunchArgsTy &LaunchArgs,
@@ -594,40 +616,20 @@ struct GenericKernelTy {
 
   /// Get the effective number of threads for the kernel based on the
   /// user-defined number of threads.
-  uint32_t getEffectiveNumThreads(GenericDeviceTy &GenericDevice,
-                                  uint32_t UserThreadLimit) const;
+  static uint32_t getEffectiveNumThreads(GenericDeviceTy &GenericDevice,
+                                         uint32_t UserThreadLimit,
+                                         const KernelLaunchArgsTy &LaunchArgs);
 
   /// Get the effective number of blocks for the kernel based on the
   /// user-defined number of blocks and the loop trip count.
   /// The number of threads \p NumThreads can be adjusted by this method.
   /// \p IsNumThreadsFromUser is true is \p NumThreads is defined by user via
   /// thread_limit clause.
-  uint32_t getEffectiveNumBlocks(GenericDeviceTy &GenericDevice,
-                                 uint32_t UserNumBlocks, uint64_t LoopTripCount,
-                                 uint32_t &EffectiveNumThreads,
-                                 bool IsNumThreadsStrict,
-                                 bool IsNumThreadsFromUser) const;
-
-  /// Indicate if the kernel works in Generic SPMD, Generic, No-Loop
-  /// or SPMD mode.
-  bool isGenericSPMDMode() const {
-    return KernelEnvironment.Configuration.ExecMode ==
-           OMP_TGT_EXEC_MODE_GENERIC_SPMD;
-  }
-  bool isGenericMode() const {
-    return KernelEnvironment.Configuration.ExecMode ==
-           OMP_TGT_EXEC_MODE_GENERIC;
-  }
-  bool isSPMDMode() const {
-    return KernelEnvironment.Configuration.ExecMode == OMP_TGT_EXEC_MODE_SPMD;
-  }
-  bool isBareMode() const {
-    return KernelEnvironment.Configuration.ExecMode == OMP_TGT_EXEC_MODE_BARE;
-  }
-  bool isNoLoopMode() const {
-    return KernelEnvironment.Configuration.ExecMode ==
-           OMP_TGT_EXEC_MODE_SPMD_NO_LOOP;
-  }
+  static uint32_t
+  getEffectiveNumBlocks(GenericDeviceTy &GenericDevice, uint32_t UserNumBlocks,
+                        uint64_t LoopTripCount, uint32_t &EffectiveNumThreads,
+                        bool IsNumThreadsStrict, bool IsNumThreadsFromUser,
+                        const KernelLaunchArgsTy &LaunchArgs);
 
   /// The kernel name.
   std::string Name;
@@ -636,18 +638,9 @@ struct GenericKernelTy {
   DeviceImageTy *ImagePtr = nullptr;
 
 protected:
-  /// The preferred number of threads to run the kernel.
-  uint32_t PreferredNumThreads;
-
-  /// The maximum number of threads which the kernel could leverage.
-  uint32_t MaxNumThreads;
-
   /// The static memory sized per block.
   uint32_t StaticBlockMemSize = 0;
 
-  /// The kernel environment, including execution flags.
-  KernelEnvironmentTy KernelEnvironment;
-
   /// The prototype kernel launch environment.
   KernelLaunchEnvironmentTy KernelLaunchEnvironment;
 };
diff --git a/offload/plugins-nextgen/common/src/PluginInterface.cpp b/offload/plugins-nextgen/common/src/PluginInterface.cpp
index e555321e1b96b..07f7789c72b93 100644
--- a/offload/plugins-nextgen/common/src/PluginInterface.cpp
+++ b/offload/plugins-nextgen/common/src/PluginInterface.cpp
@@ -72,37 +72,8 @@ void AsyncInfoWrapperTy::finalize(Error &Err) {
 
 Error GenericKernelTy::init(GenericDeviceTy &GenericDevice,
                             DeviceImageTy &Image) {
-
   ImagePtr = &Image;
 
-  // Retrieve kernel environment object for the kernel.
-  std::string EnvironmentName = std::string(Name) + "_kernel_environment";
-  GenericGlobalHandlerTy &GHandler = GenericDevice.Plugin.getGlobalHandler();
-  if (GHandler.isSymbolInImage(GenericDevice, Image, EnvironmentName)) {
-    GlobalTy KernelEnv(EnvironmentName, sizeof(KernelEnvironment),
-                       &KernelEnvironment);
-    if (auto Err =
-            GHandler.readGlobalFromImage(GenericDevice, *ImagePtr, KernelEnv))
-      return Err;
-  } else {
-    KernelEnvironment = KernelEnvironmentTy{};
-    ODBG(OLDT_Kernel) << "Failed to read kernel environment for '" << getName()
-                      << "' Using default Bare (0) execution mode";
-  }
-
-  // Max = Config.Max > 0 ? min(Config.Max, Device.Max) : Device.Max;
-  MaxNumThreads = KernelEnvironment.Configuration.MaxThreads > 0
-                      ? std::min(KernelEnvironment.Configuration.MaxThreads,
-                                 int32_t(GenericDevice.getThreadLimit()))
-                      : GenericDevice.getThreadLimit();
-
-  // Pref = Config.Pref > 0 ? max(Config.Pref, Device.Pref) : Device.Pref;
-  PreferredNumThreads =
-      KernelEnvironment.Configuration.MinThreads > 0
-          ? std::max(KernelEnvironment.Configuration.MinThreads,
-                     int32_t(GenericDevice.getDefaultNumThreads()))
-          : GenericDevice.getDefaultNumThreads();
-
   return initImpl(GenericDevice, Image);
 }
 
@@ -119,8 +90,8 @@ GenericKernelTy::getKernelLaunchEnvironment(
       !LaunchArgs.DynPtrSlot)
     return nullptr;
 
-  const auto &RedCfg = KernelEnvironment.Configuration;
-  const bool NeedsReductionBuffer = RedCfg.ReductionDataSize != 0;
+  const bool NeedsReductionBuffer =
+      LaunchArgs.KernelEnvironment.ReductionDataSize != 0;
   if (NeedsReductionBuffer && LaunchArgs.OmpABIVersion < OMP_KERNEL_ARG_VERSION)
     return Plugin::error(ErrorCode::INVALID_BINARY,
                          "kernel was built against an older OpenMP "
@@ -153,7 +124,7 @@ GenericKernelTy::getKernelLaunchEnvironment(
   if (NeedsReductionBuffer) {
     // Use number of teams many buffer elements.
     auto AllocOrErr = GenericDevice.dataAlloc(
-        uint64_t(RedCfg.ReductionDataSize) * NumBlocks0,
+        uint64_t(LaunchArgs.KernelEnvironment.ReductionDataSize) * NumBlocks0,
         /*HostPtr=*/nullptr, TargetAllocTy::TARGET_ALLOC_DEVICE,
         /*Alignment=*/0);
     if (!AllocOrErr)
@@ -186,7 +157,8 @@ Error GenericKernelTy::printLaunchInfo(GenericDeviceTy &GenericDevice,
        "Launching kernel %s with [%u,%u,%u] blocks and [%u,%u,%u] threads in "
        "%s mode\n",
        getName(), NumBlocks[0], NumBlocks[1], NumBlocks[2], NumThreads[0],
-       NumThreads[1], NumThreads[2], getExecutionModeName());
+       NumThreads[1], NumThreads[2],
+       LaunchArgs.KernelEnvironment.getExecutionModeName());
   return printLaunchInfoDetails(GenericDevice, LaunchArgs, NumThreads,
                                 NumBlocks);
 }
@@ -255,7 +227,7 @@ Error GenericKernelTy::launch(GenericDeviceTy &GenericDevice,
                                     LaunchArgs.UserNumBlocks[2]};
 
   // Multidimensional is only supported with bare mode for now.
-  assert(isBareMode() ||
+  assert(LaunchArgs.KernelEnvironment.isBareMode() ||
          EffectiveNumThreads[1] == 1 && EffectiveNumThreads[2] == 1 &&
              EffectiveNumBlocks[1] == 1 && EffectiveNumBlocks[2] == 1 &&
              "Non-bare mode should only use the first thread and block "
@@ -272,14 +244,14 @@ Error GenericKernelTy::launch(GenericDeviceTy &GenericDevice,
 
   // Calculate or adjust the effective number of threads and blocks if needed.
   if (!LaunchArgs.Flags.StrictThreads)
-    EffectiveNumThreads[0] =
-        getEffectiveNumThreads(GenericDevice, EffectiveNumThreads[0]);
+    EffectiveNumThreads[0] = getEffectiveNumThreads(
+        GenericDevice, EffectiveNumThreads[0], LaunchArgs);
 
   if (!LaunchArgs.Flags.StrictBlocks)
     EffectiveNumBlocks[0] = getEffectiveNumBlocks(
         GenericDevice, EffectiveNumBlocks[0], LaunchArgs.Tripcount,
         EffectiveNumThreads[0], LaunchArgs.Flags.StrictThreads,
-        LaunchArgs.UserThreadLimit[0] > 0);
+        LaunchArgs.UserThreadLimit[0] > 0, LaunchArgs);
 
   auto DynBlockMemConfOrErr = prepareBlockMemory(
       GenericDevice, LaunchArgs,
@@ -342,21 +314,27 @@ Error GenericKernelTy::launch(GenericDeviceTy &GenericDevice,
 
 uint32_t
 GenericKernelTy::getEffectiveNumThreads(GenericDeviceTy &GenericDevice,
-                                        uint32_t UserThreadLimit) const {
-  assert(!isBareMode() && "bare kernel should not call this function");
+                                        uint32_t UserThreadLimit,
+                                        const KernelLaunchArgsTy &LaunchArgs) {
+  assert(!LaunchArgs.KernelEnvironment.isBareMode() &&
+         "bare kernel should not call this function");
 
-  if (UserThreadLimit > 0 && isGenericMode())
+  if (UserThreadLimit > 0 && LaunchArgs.KernelEnvironment.isGenericMode())
     UserThreadLimit += GenericDevice.getWarpSize();
 
-  return std::min(MaxNumThreads, (UserThreadLimit > 0) ? UserThreadLimit
-                                                       : PreferredNumThreads);
+  return std::min(LaunchArgs.KernelEnvironment.MaxNumThreads,
+                  (UserThreadLimit > 0)
+                      ? UserThreadLimit
+                      : LaunchArgs.KernelEnvironment.PreferredNumThreads);
 }
 
 uint32_t GenericKernelTy::getEffectiveNumBlocks(
     GenericDeviceTy &GenericDevice, uint32_t UserNumBlocks,
     uint64_t LoopTripCount, uint32_t &EffectiveNumThreads,
-    bool IsNumThreadsStrict, bool IsNumThreadsFromUser) const {
-  assert(!isBareMode() && "bare kernel should not call this function");
+    bool IsNumThreadsStrict, bool IsNumThreadsFromUser,
+    const KernelLaunchArgsTy &LaunchArgs) {
+  assert(!LaunchArgs.KernelEnvironment.isBareMode() &&
+         "bare kernel should not call this function");
 
   // NOTE: This clamps the user-requested number of blocks to the device limit
   // rather than honoring it exactly, which is non-standard behavior. Truly
@@ -367,14 +345,14 @@ uint32_t GenericKernelTy::getEffectiveNumBlocks(
                     GenericDevice.getBlockLimit(EffectiveNumThreads));
 
   // Return the number of blocks required to cover the loop iterations.
-  if (isNoLoopMode())
+  if (LaunchArgs.KernelEnvironment.isNoLoopMode())
     return LoopTripCount > 0 ? (((LoopTripCount - 1) / EffectiveNumThreads) + 1)
                              : 1;
 
   uint64_t DefaultNumBlocks = GenericDevice.getDefaultNumBlocks();
   uint64_t TripCountNumBlocks = std::numeric_limits<uint64_t>::max();
   if (LoopTripCount > 0) {
-    if (isSPMDMode()) {
+    if (LaunchArgs.KernelEnvironment.isSPMDMode()) {
       // We have a combined construct, i.e. `target teams distribute
       // parallel for [simd]`. We launch so many blocks so that each thread
       // will execute one iteration of the loop; rounded up to the nearest
@@ -419,7 +397,8 @@ uint32_t GenericKernelTy::getEffectiveNumBlocks(
       assert(OldNumThreads >= EffectiveNumThreads &&
              "Number of threads cannot be increased!");
     } else {
-      assert((isGenericMode() || isGenericSPMDMode()) &&
+      assert((LaunchArgs.KernelEnvironment.isGenericMode() ||
+              LaunchArgs.KernelEnvironment.isGenericSPMDMode()) &&
              "Unexpected execution mode!");
       // If we reach this point, then we have a non-combined construct, i.e.
       // `teams distribute` with a nested `parallel for` and each block is
diff --git a/offload/plugins-nextgen/common/src/RecordReplay.cpp b/offload/plugins-nextgen/common/src/RecordReplay.cpp
index 317ebd0cc6742..768e85338a140 100644
--- a/offload/plugins-nextgen/common/src/RecordReplay.cpp
+++ b/offload/plugins-nextgen/common/src/RecordReplay.cpp
@@ -279,7 +279,7 @@ Error NativeRecordReplayTy::recordDescImpl(
 
   // Export minimum and maximum for allowed number of threads. If zero, it means
   // there was no restriction provided by the program.
-  uint32_t MaxThreads = Kernel.getMaxThreads();
+  uint32_t MaxThreads = LaunchArgs.KernelEnvironment.MaxNumThreads;
   json::Array JsonThreadsLimits;
   JsonThreadsLimits.push_back(1);
   JsonThreadsLimits.push_back(MaxThreads);
diff --git a/offload/plugins-nextgen/cuda/src/rtl.cpp b/offload/plugins-nextgen/cuda/src/rtl.cpp
index 8f1a4b29eef8b..a0335c9dfaec0 100644
--- a/offload/plugins-nextgen/cuda/src/rtl.cpp
+++ b/offload/plugins-nextgen/cuda/src/rtl.cpp
@@ -121,8 +121,8 @@ struct CUDAKernelTy : public GenericKernelTy {
     if (auto Err = Plugin::check(Res, "error in cuFuncGetAttribute: %s"))
       return Err;
 
-    // The maximum number of threads cannot exceed the maximum of the kernel.
-    MaxNumThreads = std::min(MaxNumThreads, (uint32_t)MaxThreads);
+    // The maximum number of threads per block for this kernel's function.
+    MaxNumThreads = (uint32_t)MaxThreads;
 
     int SharedMemSize;
     Res = cuFuncGetAttribute(&SharedMemSize,
@@ -162,12 +162,19 @@ struct CUDAKernelTy : public GenericKernelTy {
                               const uint32_t NumThreads[3],
                               uint32_t DynBlockMemSize) const override;
 
+  /// Return the maximum number of threads per block for this kernel's
+  /// function, as reported by the CUDA driver.
+  uint32_t getMaxThreads() const override { return MaxNumThreads; }
+
 private:
   /// The CUDA kernel function to execute.
   CUfunction Func;
   /// The maximum amount of dynamic shared memory per thread group. By default,
   /// this is set to 48 KB.
   mutable uint32_t MaxDynBlockMemSize = 49152;
+  /// The maximum number of threads per block this kernel's function may use,
+  /// as reported by the CUDA driver.
+  uint32_t MaxNumThreads = 0;
 };
 
 /// Class wrapping a CUDA stream reference. These are the objects handled by the
diff --git a/offload/plugins-nextgen/host/src/rtl.cpp b/offload/plugins-nextgen/host/src/rtl.cpp
index 55ada2f82c360..9f1aa23102ecc 100644
--- a/offload/plugins-nextgen/host/src/rtl.cpp
+++ b/offload/plugins-nextgen/host/src/rtl.cpp
@@ -76,12 +76,6 @@ struct GenELF64KernelTy : public GenericKernelTy {
     // Save the function pointer.
     Func = reinterpret_cast<KernelTy *>(Global.getPtr());
 
-    KernelEnvironment.Configuration.ExecMode = OMP_TGT_EXEC_MODE_GENERIC;
-    KernelEnvironment.Configuration.MayUseNestedParallelism = /*Unknown=*/2;
-    KernelEnvironment.Configuration.UseGenericStateMachine = /*Unknown=*/2;
-
-    // Set the maximum number of threads to a single.
-    MaxNumThreads = 1;
     return Plugin::success();
   }
 

>From 75524871d7f788811b452c84f55850e919ae5ddf Mon Sep 17 00:00:00 2001
From: "Duran, Alex" <alejandro.duran at intel.com>
Date: Thu, 10 Sep 2026 14:34:02 -0700
Subject: [PATCH 2/3] Address review comments

---
 offload/libompaccsupport/PluginManager.cpp | 18 +++++++++++-------
 1 file changed, 11 insertions(+), 7 deletions(-)

diff --git a/offload/libompaccsupport/PluginManager.cpp b/offload/libompaccsupport/PluginManager.cpp
index 5c495b666d409..babd9347655d2 100644
--- a/offload/libompaccsupport/PluginManager.cpp
+++ b/offload/libompaccsupport/PluginManager.cpp
@@ -17,6 +17,7 @@
 #include "Shared/Profile.h"
 #include "device.h"
 
+#include "llvm/ADT/SmallString.h"
 #include "llvm/Support/Error.h"
 #include "llvm/Support/ErrorHandling.h"
 #include <algorithm>
@@ -464,20 +465,25 @@ static int loadImagesOntoDevice(DeviceTy &Device) {
               REPORT() << "Failed to write symbol for USM " << Entry.SymbolName;
         } else if (Entry.Address) {
           if (Device.RTL->get_function(Binary, Entry.SymbolName,
-                                       &DeviceEntry.Address) != OFFLOAD_SUCCESS)
+                                       &DeviceEntry.Address) !=
+              OFFLOAD_SUCCESS) {
             REPORT() << "Failed to load kernel " << Entry.SymbolName;
+            continue;
+          }
 
           // Read this kernel's launch-geometry properties once, from its
           // "<name>_kernel_environment" device global, and cache them on
           // the device for use at launch time.
-          std::string EnvName =
-              std::string(Entry.SymbolName) + "_kernel_environment";
+
+          SmallString<128> EnvName(Entry.SymbolName);
+          EnvName += "_kernel_environment";
           KernelEnvironmentTy KernelEnv{};
           void *KernelEnvPtr = nullptr;
           bool ReadOk =
               Device.RTL->get_global(Binary, sizeof(KernelEnv), EnvName.c_str(),
                                      &KernelEnvPtr) == OFFLOAD_SUCCESS &&
-              Device.RTL->data_retrieve(DeviceId, &KernelEnv, KernelEnvPtr,
+              Device.RTL->data_retrieve(Device.RTLDeviceID, &KernelEnv,
+                                        KernelEnvPtr,
                                         sizeof(KernelEnv)) == OFFLOAD_SUCCESS;
           if (!ReadOk) {
             KernelEnv = KernelEnvironmentTy{};
@@ -500,7 +506,7 @@ static int loadImagesOntoDevice(DeviceTy &Device) {
           }
 
           llvm::omp::target::plugin::GenericDeviceTy &GenericDevice =
-              Device.RTL->getDevice(DeviceId);
+              Device.RTL->getDevice(Device.RTLDeviceID);
           auto *Kernel =
               reinterpret_cast<llvm::omp::target::plugin::GenericKernelTy *>(
                   DeviceEntry.Address);
@@ -518,8 +524,6 @@ static int loadImagesOntoDevice(DeviceTy &Device) {
                                       int32_t(GenericDevice.getThreadLimit()))
                            : GenericDevice.getThreadLimit(),
                        Kernel->getMaxThreads());
-          // Pref = Config.Pref > 0 ? max(Config.Pref, Device.Pref)
-          //                        : Device.Pref.
           LaunchInfo.PreferredNumThreads =
               Cfg.MinThreads > 0
                   ? std::max(Cfg.MinThreads,

>From ded3f275163e5c33c1197a5ee71e2b511a8caa02 Mon Sep 17 00:00:00 2001
From: "Duran, Alex" <alejandro.duran at intel.com>
Date: Mon, 21 Sep 2026 06:55:06 -0700
Subject: [PATCH 3/3] Remove host special handling after compiler changes

---
 offload/include/PluginManager.h            |  3 ---
 offload/libompaccsupport/PluginManager.cpp | 13 ++-----------
 2 files changed, 2 insertions(+), 14 deletions(-)

diff --git a/offload/include/PluginManager.h b/offload/include/PluginManager.h
index 7e15c68d4a916..bdfae3f0fdd93 100644
--- a/offload/include/PluginManager.h
+++ b/offload/include/PluginManager.h
@@ -150,9 +150,6 @@ struct PluginManager {
     return count;
   }
 
-  /// Return the host (GenELF64) plugin, or nullptr if it wasn't built.
-  GenericPluginTy *getHostPlugin() const { return HostPlugin; }
-
 private:
   bool RTLsLoaded = false;
   llvm::SmallVector<__tgt_bin_desc *> DelayedBinDesc;
diff --git a/offload/libompaccsupport/PluginManager.cpp b/offload/libompaccsupport/PluginManager.cpp
index babd9347655d2..db68b1c6fb9c4 100644
--- a/offload/libompaccsupport/PluginManager.cpp
+++ b/offload/libompaccsupport/PluginManager.cpp
@@ -48,8 +48,6 @@ void PluginManager::init() {
   do {                                                                         \
     Plugins.emplace_back(                                                      \
         std::unique_ptr<GenericPluginTy>(createPlugin_##Name()));              \
-    if (strcmp(#Name, "host") == 0)                                            \
-      HostPlugin = Plugins.back().get();                                       \
   } while (false);
 #include "Shared/Targets.def"
 
@@ -487,15 +485,8 @@ static int loadImagesOntoDevice(DeviceTy &Device) {
                                         sizeof(KernelEnv)) == OFFLOAD_SUCCESS;
           if (!ReadOk) {
             KernelEnv = KernelEnvironmentTy{};
-            // If no kernel environment is found, the host kernels are expected
-            // to run in Generic execution mode. Other backends expect to be run
-            // in Bare mode.
-            if (Device.RTL == PM->getHostPlugin())
-              KernelEnv.Configuration.ExecMode =
-                  llvm::omp::OMP_TGT_EXEC_MODE_GENERIC;
-            else
-              KernelEnv.Configuration.ExecMode =
-                  llvm::omp::OMP_TGT_EXEC_MODE_BARE;
+            KernelEnv.Configuration.ExecMode =
+                llvm::omp::OMP_TGT_EXEC_MODE_BARE;
             ODBG(ODT_Mapping)
                 << "Failed to read kernel environment for '" << Entry.SymbolName
                 << "', using default "



More information about the llvm-commits mailing list