[clang] [llvm] [offload][OpenMP] Improve cross-team reduction grid selection (PR #208253)
Robert Imschweiler via llvm-commits
llvm-commits at lists.llvm.org
Wed Jul 8 09:23:16 PDT 2026
https://github.com/ro-i created https://github.com/llvm/llvm-project/pull/208253
The default cross-team reduction algorithm benefits from larger and fewer teams. Implement that by using 2 x the default number of threads and 1/2 x the default number of teams for reduction kernels. This doesn't change the default total number of threads, it just redistributes them.
This is a first, rather simple heuristic, derived from (a subset of) what AOMP does.
The performance benefits I observed for the reduction tests in https://github.com/ro-i/xteam-test on a gfx942
(c71339705091500f731e2a39f247d2660bacbdce) are up to a few percent, with no regressions.
Claude assisted with this patch.
>From df6b80d3e4fd1a24c35e73bc4f4f58ca71fcec4a Mon Sep 17 00:00:00 2001
From: Robert Imschweiler <robert.imschweiler at amd.com>
Date: Wed, 8 Jul 2026 11:08:21 -0500
Subject: [PATCH] [offload][OpenMP] Improve cross-team reduction grid selection
The default cross-team reduction algorithm benefits from larger and
fewer teams. Implement that by using 2 x the default number of threads
and 1/2 x the default number of teams for reduction kernels. This
doesn't change the default total number of threads, it just
redistributes them.
This is a first, rather simple heuristic, derived from (a subset of)
what AOMP does.
The performance benefits I observed for the reduction tests in
https://github.com/ro-i/xteam-test on a gfx942
(c71339705091500f731e2a39f247d2660bacbdce) are up to a few percent, with
no regressions.
Claude assisted with this patch.
---
clang/lib/CodeGen/CGOpenMPRuntimeGPU.cpp | 22 ++++++++++++++++
.../common/include/PluginInterface.h | 5 ++++
.../common/src/PluginInterface.cpp | 26 ++++++++++++++++---
3 files changed, 49 insertions(+), 4 deletions(-)
diff --git a/clang/lib/CodeGen/CGOpenMPRuntimeGPU.cpp b/clang/lib/CodeGen/CGOpenMPRuntimeGPU.cpp
index 3fccd3a291d37..bd36758c7eea0 100644
--- a/clang/lib/CodeGen/CGOpenMPRuntimeGPU.cpp
+++ b/clang/lib/CodeGen/CGOpenMPRuntimeGPU.cpp
@@ -28,6 +28,19 @@ using namespace clang;
using namespace CodeGen;
using namespace llvm::omp;
+/// Returns true if \p D is a teams reduction directive. This corresponds to the
+/// condition under which CGOpenMPRuntimeGPU::emitReduction builds a teams
+/// reduction (its ReductionKind is OMPD_teams and there is at least one
+/// reduction clause).
+static bool hasTeamsReduction(const OMPExecutableDirective &D) {
+ if (!isOpenMPTeamsDirective(D.getDirectiveKind()))
+ return false;
+ for (const auto *C : D.getClausesOfKind<OMPReductionClause>())
+ if (C->getModifier() != OMPC_REDUCTION_inscan)
+ return true;
+ return false;
+}
+
namespace {
/// Pre(post)-action for different OpenMP constructs specialized for NVPTX.
class NVPTXActionTy final : public PrePostActionTy {
@@ -751,6 +764,15 @@ void CGOpenMPRuntimeGPU::emitKernelInit(const OMPExecutableDirective &D,
: llvm::omp::OMPTgtExecModeFlags::OMP_TGT_EXEC_MODE_GENERIC;
computeMinAndMaxThreadsAndTeams(D, CGF, Attrs);
+ // Set MaxThreads to twice the default workgroup size for SPMD cross-team
+ // reductions. For these reductions, we want to have fewer, larger teams to
+ // make better use of intra-team reduction and not over-emphasize the
+ // inter-team part. Note that this only buys us more freedom. The actual grid
+ // size for the launch is going to be chosen later by the offload plugin.
+ if (IsSPMD && Attrs.MaxThreads.front() < 0 && hasTeamsReduction(D))
+ Attrs.MaxThreads.front() =
+ 2 * CGM.getTarget().getGridValue().GV_Default_WG_Size;
+
CGBuilderTy &Bld = CGF.Builder;
Bld.restoreIP(OMPBuilder.createTargetInit(Bld, Attrs));
if (!IsSPMD)
diff --git a/offload/plugins-nextgen/common/include/PluginInterface.h b/offload/plugins-nextgen/common/include/PluginInterface.h
index dc21abf1a334a..308336e105524 100644
--- a/offload/plugins-nextgen/common/include/PluginInterface.h
+++ b/offload/plugins-nextgen/common/include/PluginInterface.h
@@ -574,6 +574,11 @@ struct GenericKernelTy {
DeviceImageTy *ImagePtr = nullptr;
protected:
+ /// Whether the kernel performs a cross-team reduction.
+ bool isCrossTeamReduction() const {
+ return KernelEnvironment.Configuration.ReductionDataSize != 0;
+ }
+
/// The preferred number of threads to run the kernel.
uint32_t PreferredNumThreads;
diff --git a/offload/plugins-nextgen/common/src/PluginInterface.cpp b/offload/plugins-nextgen/common/src/PluginInterface.cpp
index 70e10ca6be24a..6a08f9fe1853c 100644
--- a/offload/plugins-nextgen/common/src/PluginInterface.cpp
+++ b/offload/plugins-nextgen/common/src/PluginInterface.cpp
@@ -119,8 +119,7 @@ GenericKernelTy::getKernelLaunchEnvironment(
KernelArgs.Version < OMP_KERNEL_ARG_MIN_VERSION_WITH_DYN_PTR)
return nullptr;
- const auto &RedCfg = KernelEnvironment.Configuration;
- const bool NeedsReductionBuffer = RedCfg.ReductionDataSize != 0;
+ const bool NeedsReductionBuffer = isCrossTeamReduction();
if (NeedsReductionBuffer && KernelArgs.Version < OMP_KERNEL_ARG_VERSION)
return Plugin::error(ErrorCode::INVALID_BINARY,
"kernel was built against an older OpenMP "
@@ -150,6 +149,7 @@ GenericKernelTy::getKernelLaunchEnvironment(
LocalKLE.ReductionBuffer = nullptr;
if (NeedsReductionBuffer) {
+ const auto &RedCfg = KernelEnvironment.Configuration;
// Use number of teams many buffer elements.
auto AllocOrErr = GenericDevice.dataAlloc(
uint64_t(RedCfg.ReductionDataSize) * NumBlocks0,
@@ -390,8 +390,13 @@ GenericKernelTy::getEffectiveNumThreads(GenericDeviceTy &GenericDevice,
if (UserThreadLimit > 0 && isGenericMode())
UserThreadLimit += GenericDevice.getWarpSize();
- return std::min(MaxNumThreads, (UserThreadLimit > 0) ? UserThreadLimit
- : PreferredNumThreads);
+ // Favor fewer, larger teams for cross-team reductions. We deliberately
+ // increased MaxNumThreads in Codegen.
+ return std::min(
+ MaxNumThreads,
+ (UserThreadLimit > 0)
+ ? UserThreadLimit
+ : (isCrossTeamReduction() ? MaxNumThreads : PreferredNumThreads));
}
uint32_t GenericKernelTy::getEffectiveNumBlocks(
@@ -422,6 +427,19 @@ uint32_t GenericKernelTy::getEffectiveNumBlocks(
// will execute one iteration of the loop; rounded up to the nearest
// integer. However, if that results in too few blocks, we artificially
// reduce the thread count per block to increase the outer parallelism.
+
+ // Favor fewer, larger teams for cross-team reductions. As a heuristic, we
+ // aim for double the default number of threads and half the default
+ // number of blocks.
+ if (isCrossTeamReduction()) {
+ // Never launch more teams than the trip count needs.
+ uint64_t NumBlocks =
+ std::min<uint64_t>(std::max<uint64_t>(DefaultNumBlocks / 2, 1),
+ ((LoopTripCount - 1) / EffectiveNumThreads) + 1);
+ return std::min(NumBlocks, uint64_t(GenericDevice.getBlockLimit(
+ EffectiveNumThreads)));
+ }
+
auto MinThreads = GenericDevice.getMinThreadsForLowTripCountLoop();
MinThreads = std::min(MinThreads, EffectiveNumThreads);
More information about the llvm-commits
mailing list