[llvm] [OpenMPOpt][PGO] Set entry count on merged parallel wrapper (PR #221654)
Alok Kumar Sharma via llvm-commits
llvm-commits at lists.llvm.org
Wed Sep 30 23:51:39 PDT 2026
https://github.com/alokkrsharma updated https://github.com/llvm/llvm-project/pull/221654
>From ee811d0ddd96f402af6437b9cd59f7006bfc752a Mon Sep 17 00:00:00 2001
From: Alok Kumar Sharma <AlokKumar.Sharma at amd.com>
Date: Mon, 7 Sep 2026 13:20:04 +0530
Subject: [PATCH 1/4] [OpenMPOpt][PGO] Set entry count on merged parallel
wrapper
The outlined wrapper created by parallel-region merging has no
function_entry_count. Copy the largest count from the original outlined
callbacks so the merged region stays profiled.
---
llvm/lib/Transforms/IPO/OpenMPOpt.cpp | 23 +++
.../OpenMP/par_region_merge_profile.ll | 146 ++++++++++++++++++
2 files changed, 169 insertions(+)
create mode 100644 llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
diff --git a/llvm/lib/Transforms/IPO/OpenMPOpt.cpp b/llvm/lib/Transforms/IPO/OpenMPOpt.cpp
index 7bf628ce27fa88b..5d11a827eaebfc8 100644
--- a/llvm/lib/Transforms/IPO/OpenMPOpt.cpp
+++ b/llvm/lib/Transforms/IPO/OpenMPOpt.cpp
@@ -998,6 +998,28 @@ struct OffloadArray {
}
};
+/// Copy the max outlined-callback entry count onto the merged wrapper.
+/// \p CallbackOpNo is the callback argument (2 for __kmpc_fork_call's
+/// microtask).
+static void setMergedWrapperEntryCount(Function &WrapperFn,
+ ArrayRef<CallInst *> ForkCalls,
+ unsigned CallbackOpNo = 2) {
+ std::optional<uint64_t> EntryCount;
+ for (CallInst *CI : ForkCalls) {
+ auto *Callback = dyn_cast<Function>(
+ CI->getArgOperand(CallbackOpNo)->stripPointerCasts());
+ if (!Callback)
+ continue;
+ if (std::optional<uint64_t> EC = Callback->getEntryCount())
+ // Each callback runs once per wrapper entry, so the counts should match.
+ // The Sample profiles can disagree slightly, so take the largest.
+ EntryCount = EntryCount ? std::max(*EntryCount, *EC) : *EC;
+ }
+ // Leave the wrapper unprofiled if none of the callbacks have a count.
+ if (EntryCount)
+ WrapperFn.setEntryCount(*EntryCount);
+}
+
struct OpenMPOpt {
using OptimizationRemarkGetter =
@@ -1341,6 +1363,7 @@ struct OpenMPOpt {
OMPInfoCache.OMPBuilder.finalize(OriginalFn);
Function *OutlinedFn = MergableCIs.front()->getCaller();
+ setMergedWrapperEntryCount(*OutlinedFn, MergableCIs);
// Replace the __kmpc_fork_call calls with direct calls to the outlined
// callbacks.
diff --git a/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll b/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
new file mode 100644
index 000000000000000..2df3aa5c09270f8
--- /dev/null
+++ b/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
@@ -0,0 +1,146 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --include-generated-funcs --version 6
+; RUN: opt -S -passes=openmp-opt-cgscc -openmp-opt-enable-merging < %s | FileCheck %s
+
+; Merged wrapper should get the max outlined function_entry_count (435).
+
+%struct.ident_t = type { i32, i32, i32, i32, ptr }
+
+ at 0 = private unnamed_addr constant [23 x i8] c";unknown;unknown;0;0;;\00", align 1
+ at 1 = private unnamed_addr constant %struct.ident_t { i32 0, i32 2, i32 0, i32 22, ptr @0 }, align 8
+
+declare void @use(i32 noundef) local_unnamed_addr
+
+define dso_local void @merge_profile(i32 noundef %a, i32 noundef %cond) local_unnamed_addr !prof !47 {
+entry:
+ %a.addr = alloca i32, align 4
+ %cond.addr = alloca i32, align 4
+ store i32 %a, ptr %a.addr, align 4
+ store i32 %cond, ptr %cond.addr, align 4
+ call void (ptr, i32, ptr, ...) @__kmpc_fork_call(ptr nonnull @1, i32 2, ptr nonnull @merge_profile.omp_outlined, ptr nonnull %cond.addr, ptr nonnull %a.addr)
+ call void (ptr, i32, ptr, ...) @__kmpc_fork_call(ptr nonnull @1, i32 2, ptr nonnull @merge_profile.omp_outlined.1, ptr nonnull %cond.addr, ptr nonnull %a.addr)
+ ret void
+}
+
+define internal void @merge_profile.omp_outlined(ptr noalias readnone captures(none) %0, ptr noalias readnone captures(none) %1, ptr noundef nonnull readonly align 4 dereferenceable(4) %cond, ptr noundef nonnull readonly align 4 dereferenceable(4) %a) !prof !48 {
+entry:
+ %v = load i32, ptr %a, align 4
+ tail call void @use(i32 noundef %v)
+ ret void
+}
+
+define internal void @merge_profile.omp_outlined.1(ptr noalias readnone captures(none) %0, ptr noalias readnone captures(none) %1, ptr noundef nonnull readonly align 4 dereferenceable(4) %cond, ptr noundef nonnull readonly align 4 dereferenceable(4) %a) !prof !52 {
+entry:
+ %c = load i32, ptr %cond, align 4
+ %t = icmp eq i32 %c, 0
+ %v = load i32, ptr %a, align 4
+ %off = select i1 %t, i32 3, i32 2, !prof !54
+ %sum = add nsw i32 %v, %off
+ tail call void @use(i32 noundef %sum)
+ ret void
+}
+
+declare !callback !50 void @__kmpc_fork_call(ptr, i32, ptr, ...) local_unnamed_addr
+declare i32 @__kmpc_global_thread_num(ptr) local_unnamed_addr
+declare void @__kmpc_barrier(ptr, i32) local_unnamed_addr
+
+!llvm.module.flags = !{!0, !4}
+!0 = !{i32 7, !"openmp", i32 51}
+!4 = !{i32 1, !"ProfileSummary", !5}
+!5 = !{!6, !7, !8, !9, !10, !11, !12, !13}
+!6 = !{!"ProfileFormat", !"InstrProf"}
+!7 = !{!"TotalCount", i64 827}
+!8 = !{!"MaxCount", i64 435}
+!9 = !{!"MaxInternalCount", i64 205}
+!10 = !{!"MaxFunctionCount", i64 435}
+!11 = !{!"NumCounts", i64 4}
+!12 = !{!"NumFunctions", i64 3}
+!13 = !{!"DetailedSummary", !14}
+!14 = !{!15}
+!15 = !{i32 10000, i64 435, i32 1}
+!47 = !{!"function_entry_count", i64 2}
+!48 = !{!"function_entry_count", i64 435}
+!50 = !{!51}
+!51 = !{i64 2, i64 -1, i64 -1, i1 true}
+!52 = !{!"function_entry_count", i64 392}
+!54 = !{!"branch_weights", i32 205, i32 187}
+
+; CHECK-LABEL: define dso_local void @merge_profile(
+; CHECK-SAME: i32 noundef [[A:%.*]], i32 noundef [[COND:%.*]]) local_unnamed_addr !prof [[PROF13:![0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[STRUCTARG:%.*]] = alloca { ptr, ptr }, align 8
+; CHECK-NEXT: [[A_ADDR:%.*]] = alloca i32, align 4
+; CHECK-NEXT: [[COND_ADDR:%.*]] = alloca i32, align 4
+; CHECK-NEXT: store i32 [[A]], ptr [[A_ADDR]], align 4
+; CHECK-NEXT: store i32 [[COND]], ptr [[COND_ADDR]], align 4
+; CHECK-NEXT: br label %[[OMP_PARALLEL:.*]]
+; CHECK: [[OMP_PARALLEL]]:
+; CHECK-NEXT: [[GEP_A_ADDR:%.*]] = getelementptr { ptr, ptr }, ptr [[STRUCTARG]], i32 0, i32 0
+; CHECK-NEXT: store ptr [[COND_ADDR]], ptr [[GEP_A_ADDR]], align 8
+; CHECK-NEXT: [[GEP_COND_ADDR:%.*]] = getelementptr { ptr, ptr }, ptr [[STRUCTARG]], i32 0, i32 1
+; CHECK-NEXT: store ptr [[A_ADDR]], ptr [[GEP_COND_ADDR]], align 8
+; CHECK-NEXT: call void (ptr, i32, ptr, ...) @__kmpc_fork_call(ptr @[[GLOB1:[0-9]+]], i32 1, ptr @merge_profile..omp_par, ptr [[STRUCTARG]])
+; CHECK-NEXT: br label %[[OMP_PAR_EXIT:.*]]
+; CHECK: [[OMP_PAR_EXIT]]:
+; CHECK-NEXT: br label %[[ENTRY_SPLIT_SPLIT:.*]]
+; CHECK: [[ENTRY_SPLIT_SPLIT]]:
+; CHECK-NEXT: ret void
+;
+;
+; CHECK-LABEL: define internal void @merge_profile..omp_par(
+; CHECK-SAME: ptr noalias [[TID_ADDR:%.*]], ptr noalias [[ZERO_ADDR:%.*]], ptr [[TMP0:%.*]]) #[[ATTR0:[0-9]+]] !prof [[PROF14:![0-9]+]] {
+; CHECK-NEXT: [[OMP_PAR_ENTRY:.*:]]
+; CHECK-NEXT: [[GEP_A_ADDR:%.*]] = getelementptr { ptr, ptr }, ptr [[TMP0]], i32 0, i32 0
+; CHECK-NEXT: [[LOADGEP_COND_ADDR:%.*]] = load ptr, ptr [[GEP_A_ADDR]], align 8, !align [[META15:![0-9]+]]
+; CHECK-NEXT: [[GEP_COND_ADDR:%.*]] = getelementptr { ptr, ptr }, ptr [[TMP0]], i32 0, i32 1
+; CHECK-NEXT: [[LOADGEP_A_ADDR:%.*]] = load ptr, ptr [[GEP_COND_ADDR]], align 8, !align [[META15]]
+; CHECK-NEXT: [[TID_ADDR_LOCAL:%.*]] = alloca i32, align 4
+; CHECK-NEXT: [[TMP1:%.*]] = load i32, ptr [[TID_ADDR]], align 4
+; CHECK-NEXT: store i32 [[TMP1]], ptr [[TID_ADDR_LOCAL]], align 4
+; CHECK-NEXT: [[TID:%.*]] = load i32, ptr [[TID_ADDR_LOCAL]], align 4
+; CHECK-NEXT: br label %[[OMP_PAR_REGION:.*]]
+; CHECK: [[OMP_PAR_REGION]]:
+; CHECK-NEXT: br label %[[OMP_PAR_MERGED:.*]]
+; CHECK: [[OMP_PAR_MERGED]]:
+; CHECK-NEXT: call void (ptr, ptr, ...) @merge_profile.omp_outlined(ptr [[TID_ADDR]], ptr [[ZERO_ADDR]], ptr nonnull [[LOADGEP_COND_ADDR]], ptr nonnull [[LOADGEP_A_ADDR]])
+; CHECK-NEXT: [[OMP_GLOBAL_THREAD_NUM:%.*]] = call i32 @__kmpc_global_thread_num(ptr @[[GLOB1]])
+; CHECK-NEXT: call void @__kmpc_barrier(ptr @[[GLOB2:[0-9]+]], i32 [[OMP_GLOBAL_THREAD_NUM]])
+; CHECK-NEXT: call void (ptr, ptr, ...) @merge_profile.omp_outlined.1(ptr [[TID_ADDR]], ptr [[ZERO_ADDR]], ptr nonnull [[LOADGEP_COND_ADDR]], ptr nonnull [[LOADGEP_A_ADDR]])
+; CHECK-NEXT: br label %[[OMP_PAR_REGION_SPLIT:.*]]
+; CHECK: [[OMP_PAR_REGION_SPLIT]]:
+; CHECK-NEXT: br label %[[OMP_PAR_PRE_FINALIZE:.*]]
+; CHECK: [[OMP_PAR_PRE_FINALIZE]]:
+; CHECK-NEXT: br label %[[DOTFINI:.*]]
+; CHECK: [[DOTFINI]]:
+; CHECK-NEXT: br [[OMP_PAR_EXIT_EXITSTUB:label %.*]]
+; CHECK: [[_FINI:.*:]]
+; CHECK-NEXT: br label %[[OMP_PAR_EXIT_EXITSTUB1:.*]]
+; CHECK: [[OMP_PAR_EXIT_EXITSTUB1]]:
+; CHECK-NEXT: ret void
+;
+;
+; CHECK-LABEL: define internal void @merge_profile.omp_outlined(
+; CHECK-SAME: ptr noalias readnone captures(none) [[TMP0:%.*]], ptr noalias readnone captures(none) [[TMP1:%.*]], ptr noundef nonnull readonly align 4 dereferenceable(4) [[COND:%.*]], ptr noundef nonnull readonly align 4 dereferenceable(4) [[A:%.*]]) !prof [[PROF14]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[A]], align 4
+; CHECK-NEXT: tail call void @use(i32 noundef [[V]])
+; CHECK-NEXT: ret void
+;
+;
+; CHECK-LABEL: define internal void @merge_profile.omp_outlined.1(
+; CHECK-SAME: ptr noalias readnone captures(none) [[TMP0:%.*]], ptr noalias readnone captures(none) [[TMP1:%.*]], ptr noundef nonnull readonly align 4 dereferenceable(4) [[COND:%.*]], ptr noundef nonnull readonly align 4 dereferenceable(4) [[A:%.*]]) !prof [[PROF16:![0-9]+]] {
+; CHECK-NEXT: [[ENTRY:.*:]]
+; CHECK-NEXT: [[C:%.*]] = load i32, ptr [[COND]], align 4
+; CHECK-NEXT: [[T:%.*]] = icmp eq i32 [[C]], 0
+; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[A]], align 4
+; CHECK-NEXT: [[OFF:%.*]] = select i1 [[T]], i32 3, i32 2, !prof [[PROF17:![0-9]+]]
+; CHECK-NEXT: [[SUM:%.*]] = add nsw i32 [[V]], [[OFF]]
+; CHECK-NEXT: tail call void @use(i32 noundef [[SUM]])
+; CHECK-NEXT: ret void
+;
+;.
+; CHECK: [[PROF13]] = !{!"function_entry_count", i64 2}
+; CHECK: [[PROF14]] = !{!"function_entry_count", i64 435}
+; CHECK: [[META15]] = !{i64 4}
+; CHECK: [[PROF16]] = !{!"function_entry_count", i64 392}
+; CHECK: [[PROF17]] = !{!"branch_weights", i32 205, i32 187}
+;.
>From 3b709091689851ca6974602b039e62fd20b738aa Mon Sep 17 00:00:00 2001
From: Alok Kumar Sharma <AlokKumar.Sharma at amd.com>
Date: Fri, 25 Sep 2026 15:43:19 +0530
Subject: [PATCH 2/4] Review comments
---
.../OpenMP/par_region_merge_profile.ll | 54 ++++++++++---------
1 file changed, 28 insertions(+), 26 deletions(-)
diff --git a/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll b/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
index 2df3aa5c09270f8..6c040d919060285 100644
--- a/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
+++ b/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
@@ -10,7 +10,7 @@
declare void @use(i32 noundef) local_unnamed_addr
-define dso_local void @merge_profile(i32 noundef %a, i32 noundef %cond) local_unnamed_addr !prof !47 {
+define dso_local void @merge_profile(i32 noundef %a, i32 noundef %cond) local_unnamed_addr !prof !13 {
entry:
%a.addr = alloca i32, align 4
%cond.addr = alloca i32, align 4
@@ -21,48 +21,50 @@ entry:
ret void
}
-define internal void @merge_profile.omp_outlined(ptr noalias readnone captures(none) %0, ptr noalias readnone captures(none) %1, ptr noundef nonnull readonly align 4 dereferenceable(4) %cond, ptr noundef nonnull readonly align 4 dereferenceable(4) %a) !prof !48 {
+define internal void @merge_profile.omp_outlined(ptr noalias readnone captures(none) %0, ptr noalias readnone captures(none) %1, ptr noundef nonnull readonly align 4 dereferenceable(4) %cond, ptr noundef nonnull readonly align 4 dereferenceable(4) %a) !prof !14 {
entry:
%v = load i32, ptr %a, align 4
tail call void @use(i32 noundef %v)
ret void
}
-define internal void @merge_profile.omp_outlined.1(ptr noalias readnone captures(none) %0, ptr noalias readnone captures(none) %1, ptr noundef nonnull readonly align 4 dereferenceable(4) %cond, ptr noundef nonnull readonly align 4 dereferenceable(4) %a) !prof !52 {
+define internal void @merge_profile.omp_outlined.1(ptr noalias readnone captures(none) %0, ptr noalias readnone captures(none) %1, ptr noundef nonnull readonly align 4 dereferenceable(4) %cond, ptr noundef nonnull readonly align 4 dereferenceable(4) %a) !prof !15 {
entry:
%c = load i32, ptr %cond, align 4
%t = icmp eq i32 %c, 0
%v = load i32, ptr %a, align 4
- %off = select i1 %t, i32 3, i32 2, !prof !54
+ %off = select i1 %t, i32 3, i32 2, !prof !16
%sum = add nsw i32 %v, %off
tail call void @use(i32 noundef %sum)
ret void
}
-declare !callback !50 void @__kmpc_fork_call(ptr, i32, ptr, ...) local_unnamed_addr
+declare !callback !17 void @__kmpc_fork_call(ptr, i32, ptr, ...) local_unnamed_addr
+
declare i32 @__kmpc_global_thread_num(ptr) local_unnamed_addr
declare void @__kmpc_barrier(ptr, i32) local_unnamed_addr
-!llvm.module.flags = !{!0, !4}
+!llvm.module.flags = !{!0, !1}
+
!0 = !{i32 7, !"openmp", i32 51}
-!4 = !{i32 1, !"ProfileSummary", !5}
-!5 = !{!6, !7, !8, !9, !10, !11, !12, !13}
-!6 = !{!"ProfileFormat", !"InstrProf"}
-!7 = !{!"TotalCount", i64 827}
-!8 = !{!"MaxCount", i64 435}
-!9 = !{!"MaxInternalCount", i64 205}
-!10 = !{!"MaxFunctionCount", i64 435}
-!11 = !{!"NumCounts", i64 4}
-!12 = !{!"NumFunctions", i64 3}
-!13 = !{!"DetailedSummary", !14}
-!14 = !{!15}
-!15 = !{i32 10000, i64 435, i32 1}
-!47 = !{!"function_entry_count", i64 2}
-!48 = !{!"function_entry_count", i64 435}
-!50 = !{!51}
-!51 = !{i64 2, i64 -1, i64 -1, i1 true}
-!52 = !{!"function_entry_count", i64 392}
-!54 = !{!"branch_weights", i32 205, i32 187}
+!1 = !{i32 1, !"ProfileSummary", !2}
+!2 = !{!3, !4, !5, !6, !7, !8, !9, !10}
+!3 = !{!"ProfileFormat", !"InstrProf"}
+!4 = !{!"TotalCount", i64 827}
+!5 = !{!"MaxCount", i64 435}
+!6 = !{!"MaxInternalCount", i64 205}
+!7 = !{!"MaxFunctionCount", i64 435}
+!8 = !{!"NumCounts", i64 4}
+!9 = !{!"NumFunctions", i64 3}
+!10 = !{!"DetailedSummary", !11}
+!11 = !{!12}
+!12 = !{i32 10000, i64 435, i32 1}
+!13 = !{!"function_entry_count", i64 2}
+!14 = !{!"function_entry_count", i64 435}
+!15 = !{!"function_entry_count", i64 392}
+!16 = !{!"branch_weights", i32 205, i32 187}
+!17 = !{!18}
+!18 = !{i64 2, i64 -1, i64 -1, i1 true}
; CHECK-LABEL: define dso_local void @merge_profile(
; CHECK-SAME: i32 noundef [[A:%.*]], i32 noundef [[COND:%.*]]) local_unnamed_addr !prof [[PROF13:![0-9]+]] {
@@ -111,8 +113,8 @@ declare void @__kmpc_barrier(ptr, i32) local_unnamed_addr
; CHECK: [[OMP_PAR_PRE_FINALIZE]]:
; CHECK-NEXT: br label %[[DOTFINI:.*]]
; CHECK: [[DOTFINI]]:
-; CHECK-NEXT: br [[OMP_PAR_EXIT_EXITSTUB:label %.*]]
-; CHECK: [[_FINI:.*:]]
+; CHECK-NEXT: br label %[[OMP_PAR_EXIT_EXITSTUB:.*]]
+; CHECK: [[OMP_PAR_EXIT_EXITSTUB]]:
; CHECK-NEXT: br label %[[OMP_PAR_EXIT_EXITSTUB1:.*]]
; CHECK: [[OMP_PAR_EXIT_EXITSTUB1]]:
; CHECK-NEXT: ret void
>From c4c5bbadcc852e2193684448ec13a6773605bef6 Mon Sep 17 00:00:00 2001
From: alosharm <AlokKumar.Sharma at amd.com>
Date: Thu, 1 Oct 2026 11:34:35 +0530
Subject: [PATCH 3/4] Review comments.
---
llvm/lib/Transforms/IPO/OpenMPOpt.cpp | 7 ++++---
llvm/test/Transforms/OpenMP/par_region_merge_profile.ll | 9 +++++++--
2 files changed, 11 insertions(+), 5 deletions(-)
diff --git a/llvm/lib/Transforms/IPO/OpenMPOpt.cpp b/llvm/lib/Transforms/IPO/OpenMPOpt.cpp
index 5d11a827eaebfc8..1c87735c4a8e708 100644
--- a/llvm/lib/Transforms/IPO/OpenMPOpt.cpp
+++ b/llvm/lib/Transforms/IPO/OpenMPOpt.cpp
@@ -999,11 +999,11 @@ struct OffloadArray {
};
/// Copy the max outlined-callback entry count onto the merged wrapper.
-/// \p CallbackOpNo is the callback argument (2 for __kmpc_fork_call's
+/// \p CallbackOpNo is the callback argument (for __kmpc_fork_call's
/// microtask).
static void setMergedWrapperEntryCount(Function &WrapperFn,
ArrayRef<CallInst *> ForkCalls,
- unsigned CallbackOpNo = 2) {
+ unsigned CallbackOpNo) {
std::optional<uint64_t> EntryCount;
for (CallInst *CI : ForkCalls) {
auto *Callback = dyn_cast<Function>(
@@ -1363,7 +1363,8 @@ struct OpenMPOpt {
OMPInfoCache.OMPBuilder.finalize(OriginalFn);
Function *OutlinedFn = MergableCIs.front()->getCaller();
- setMergedWrapperEntryCount(*OutlinedFn, MergableCIs);
+ setMergedWrapperEntryCount(*OutlinedFn, MergableCIs,
+ CallbackCalleeOperand);
// Replace the __kmpc_fork_call calls with direct calls to the outlined
// callbacks.
diff --git a/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll b/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
index 6c040d919060285..cd63d9c8aab8556 100644
--- a/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
+++ b/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
@@ -1,7 +1,12 @@
; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --include-generated-funcs --version 6
; RUN: opt -S -passes=openmp-opt-cgscc -openmp-opt-enable-merging < %s | FileCheck %s
-; Merged wrapper should get the max outlined function_entry_count (435).
+; OpenMPOpt parallel-region merging with sample PGO (!prof metadata).
+; Verifies that merging preserves profile metadata on the original outlined
+; callbacks and gives the merged wrapper (@merge_profile..omp_par) the
+; maximum function_entry_count of the merged callbacks.
+; The callback counts intentionally differ (435 vs 392). Since sample PGO
+; profiles are approximate, unlike InstrProf, the wrapper uses the maximum.
%struct.ident_t = type { i32, i32, i32, i32, ptr }
@@ -49,7 +54,7 @@ declare void @__kmpc_barrier(ptr, i32) local_unnamed_addr
!0 = !{i32 7, !"openmp", i32 51}
!1 = !{i32 1, !"ProfileSummary", !2}
!2 = !{!3, !4, !5, !6, !7, !8, !9, !10}
-!3 = !{!"ProfileFormat", !"InstrProf"}
+!3 = !{!"ProfileFormat", !"SampleProfile"}
!4 = !{!"TotalCount", i64 827}
!5 = !{!"MaxCount", i64 435}
!6 = !{!"MaxInternalCount", i64 205}
>From e825c9a93ddafc4b4ea43418deded7c5457fad01 Mon Sep 17 00:00:00 2001
From: alosharm <AlokKumar.Sharma at amd.com>
Date: Thu, 1 Oct 2026 11:43:53 +0530
Subject: [PATCH 4/4] [OpenMPOpt][PGO] Preserve callsite counts in merged
parallel wrappers
Attach the merged wrapper's entry count to each outlined callback call.
Without a callsite weight, sample PGO treats these calls as cold once
the wrapper has profile data.
---
llvm/lib/Transforms/IPO/OpenMPOpt.cpp | 46 ++++++++++++++-----
.../OpenMP/par_region_merge_profile.ll | 15 +++---
2 files changed, 44 insertions(+), 17 deletions(-)
diff --git a/llvm/lib/Transforms/IPO/OpenMPOpt.cpp b/llvm/lib/Transforms/IPO/OpenMPOpt.cpp
index 1c87735c4a8e708..5fcf857d8047546 100644
--- a/llvm/lib/Transforms/IPO/OpenMPOpt.cpp
+++ b/llvm/lib/Transforms/IPO/OpenMPOpt.cpp
@@ -52,6 +52,8 @@
#include "llvm/IR/IntrinsicsNVPTX.h"
#include "llvm/IR/LLVMContext.h"
#include "llvm/IR/MDBuilder.h"
+#include "llvm/IR/ProfDataUtils.h"
+#include "llvm/IR/ProfileSummary.h"
#include "llvm/Support/Casting.h"
#include "llvm/Support/CommandLine.h"
#include "llvm/Support/Debug.h"
@@ -60,6 +62,8 @@
#include "llvm/Transforms/Utils/CallGraphUpdater.h"
#include <algorithm>
+#include <limits>
+#include <memory>
#include <optional>
#include <string>
@@ -998,12 +1002,12 @@ struct OffloadArray {
}
};
-/// Copy the max outlined-callback entry count onto the merged wrapper.
-/// \p CallbackOpNo is the callback argument (for __kmpc_fork_call's
-/// microtask).
-static void setMergedWrapperEntryCount(Function &WrapperFn,
- ArrayRef<CallInst *> ForkCalls,
- unsigned CallbackOpNo) {
+// Use the max outlined entry count. Instrumentation counts should already
+// match across merged callbacks, but sample profiles can differ. Returns
+// nullopt when no callback entry count is available.
+static std::optional<uint64_t>
+getMergedWrapperEntryCount(ArrayRef<CallInst *> ForkCalls,
+ unsigned CallbackOpNo) {
std::optional<uint64_t> EntryCount;
for (CallInst *CI : ForkCalls) {
auto *Callback = dyn_cast<Function>(
@@ -1015,9 +1019,13 @@ static void setMergedWrapperEntryCount(Function &WrapperFn,
// The Sample profiles can disagree slightly, so take the largest.
EntryCount = EntryCount ? std::max(*EntryCount, *EC) : *EC;
}
- // Leave the wrapper unprofiled if none of the callbacks have a count.
- if (EntryCount)
- WrapperFn.setEntryCount(*EntryCount);
+ return EntryCount;
+}
+
+static bool moduleHasSampleProfile(const Module &M) {
+ std::unique_ptr<ProfileSummary> Summary(
+ ProfileSummary::getFromMD(M.getProfileSummary(/*IsCS=*/false)));
+ return Summary && Summary->getKind() == ProfileSummary::PSK_Sample;
}
struct OpenMPOpt {
@@ -1363,8 +1371,15 @@ struct OpenMPOpt {
OMPInfoCache.OMPBuilder.finalize(OriginalFn);
Function *OutlinedFn = MergableCIs.front()->getCaller();
- setMergedWrapperEntryCount(*OutlinedFn, MergableCIs,
- CallbackCalleeOperand);
+ std::optional<uint64_t> WrapperCount =
+ getMergedWrapperEntryCount(MergableCIs, CallbackCalleeOperand);
+ // Leave the wrapper unprofiled when no callback has an entry count.
+ if (WrapperCount)
+ OutlinedFn->setEntryCount(*WrapperCount);
+ // Only sample PGO treats a profiled caller with no callsite weight as
+ // cold. Instrumentation profiles derive that count from the entry count.
+ const bool SampleProfile =
+ moduleHasSampleProfile(*OriginalFn->getParent());
// Replace the __kmpc_fork_call calls with direct calls to the outlined
// callbacks.
@@ -1383,6 +1398,15 @@ struct OpenMPOpt {
CallInst::Create(FT, Callee, Args, "", CI->getIterator());
if (CI->getDebugLoc())
NewCI->setDebugLoc(CI->getDebugLoc());
+ // Each body runs once per wrapper entry. Without a callsite weight,
+ // sample PGO treats these calls as cold.
+ if (WrapperCount && SampleProfile) {
+ uint64_t Count = *WrapperCount;
+ uint32_t Weight = Count > std::numeric_limits<uint32_t>::max()
+ ? std::numeric_limits<uint32_t>::max()
+ : static_cast<uint32_t>(Count);
+ setBranchWeights(*NewCI, {Weight}, /*IsExpected=*/false);
+ }
// Forward parameter attributes from the callback to the callee.
for (unsigned U = CallbackFirstArgOperand, E = CI->arg_size(); U < E;
diff --git a/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll b/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
index cd63d9c8aab8556..374c8758d129d21 100644
--- a/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
+++ b/llvm/test/Transforms/OpenMP/par_region_merge_profile.ll
@@ -7,6 +7,8 @@
; maximum function_entry_count of the merged callbacks.
; The callback counts intentionally differ (435 vs 392). Since sample PGO
; profiles are approximate, unlike InstrProf, the wrapper uses the maximum.
+; Both direct calls get that count too, otherwise sample PGO treats them
+; as cold.
%struct.ident_t = type { i32, i32, i32, i32, ptr }
@@ -108,10 +110,10 @@ declare void @__kmpc_barrier(ptr, i32) local_unnamed_addr
; CHECK: [[OMP_PAR_REGION]]:
; CHECK-NEXT: br label %[[OMP_PAR_MERGED:.*]]
; CHECK: [[OMP_PAR_MERGED]]:
-; CHECK-NEXT: call void (ptr, ptr, ...) @merge_profile.omp_outlined(ptr [[TID_ADDR]], ptr [[ZERO_ADDR]], ptr nonnull [[LOADGEP_COND_ADDR]], ptr nonnull [[LOADGEP_A_ADDR]])
+; CHECK-NEXT: call void (ptr, ptr, ...) @merge_profile.omp_outlined(ptr [[TID_ADDR]], ptr [[ZERO_ADDR]], ptr nonnull [[LOADGEP_COND_ADDR]], ptr nonnull [[LOADGEP_A_ADDR]]), !prof [[PROF16:![0-9]+]]
; CHECK-NEXT: [[OMP_GLOBAL_THREAD_NUM:%.*]] = call i32 @__kmpc_global_thread_num(ptr @[[GLOB1]])
; CHECK-NEXT: call void @__kmpc_barrier(ptr @[[GLOB2:[0-9]+]], i32 [[OMP_GLOBAL_THREAD_NUM]])
-; CHECK-NEXT: call void (ptr, ptr, ...) @merge_profile.omp_outlined.1(ptr [[TID_ADDR]], ptr [[ZERO_ADDR]], ptr nonnull [[LOADGEP_COND_ADDR]], ptr nonnull [[LOADGEP_A_ADDR]])
+; CHECK-NEXT: call void (ptr, ptr, ...) @merge_profile.omp_outlined.1(ptr [[TID_ADDR]], ptr [[ZERO_ADDR]], ptr nonnull [[LOADGEP_COND_ADDR]], ptr nonnull [[LOADGEP_A_ADDR]]), !prof [[PROF16]]
; CHECK-NEXT: br label %[[OMP_PAR_REGION_SPLIT:.*]]
; CHECK: [[OMP_PAR_REGION_SPLIT]]:
; CHECK-NEXT: br label %[[OMP_PAR_PRE_FINALIZE:.*]]
@@ -134,12 +136,12 @@ declare void @__kmpc_barrier(ptr, i32) local_unnamed_addr
;
;
; CHECK-LABEL: define internal void @merge_profile.omp_outlined.1(
-; CHECK-SAME: ptr noalias readnone captures(none) [[TMP0:%.*]], ptr noalias readnone captures(none) [[TMP1:%.*]], ptr noundef nonnull readonly align 4 dereferenceable(4) [[COND:%.*]], ptr noundef nonnull readonly align 4 dereferenceable(4) [[A:%.*]]) !prof [[PROF16:![0-9]+]] {
+; CHECK-SAME: ptr noalias readnone captures(none) [[TMP0:%.*]], ptr noalias readnone captures(none) [[TMP1:%.*]], ptr noundef nonnull readonly align 4 dereferenceable(4) [[COND:%.*]], ptr noundef nonnull readonly align 4 dereferenceable(4) [[A:%.*]]) !prof [[PROF17:![0-9]+]] {
; CHECK-NEXT: [[ENTRY:.*:]]
; CHECK-NEXT: [[C:%.*]] = load i32, ptr [[COND]], align 4
; CHECK-NEXT: [[T:%.*]] = icmp eq i32 [[C]], 0
; CHECK-NEXT: [[V:%.*]] = load i32, ptr [[A]], align 4
-; CHECK-NEXT: [[OFF:%.*]] = select i1 [[T]], i32 3, i32 2, !prof [[PROF17:![0-9]+]]
+; CHECK-NEXT: [[OFF:%.*]] = select i1 [[T]], i32 3, i32 2, !prof [[PROF18:![0-9]+]]
; CHECK-NEXT: [[SUM:%.*]] = add nsw i32 [[V]], [[OFF]]
; CHECK-NEXT: tail call void @use(i32 noundef [[SUM]])
; CHECK-NEXT: ret void
@@ -148,6 +150,7 @@ declare void @__kmpc_barrier(ptr, i32) local_unnamed_addr
; CHECK: [[PROF13]] = !{!"function_entry_count", i64 2}
; CHECK: [[PROF14]] = !{!"function_entry_count", i64 435}
; CHECK: [[META15]] = !{i64 4}
-; CHECK: [[PROF16]] = !{!"function_entry_count", i64 392}
-; CHECK: [[PROF17]] = !{!"branch_weights", i32 205, i32 187}
+; CHECK: [[PROF16]] = !{!"branch_weights", i32 435}
+; CHECK: [[PROF17]] = !{!"function_entry_count", i64 392}
+; CHECK: [[PROF18]] = !{!"branch_weights", i32 205, i32 187}
;.
More information about the llvm-commits
mailing list