[llvm] [OpenMPOpt][PGO] Set entry count on merged parallel wrapper (PR #221654)

Alok Kumar Sharma via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 30 23:48:46 PDT 2026


================
@@ -0,0 +1,148 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --include-generated-funcs --version 6
+; RUN: opt -S -passes=openmp-opt-cgscc -openmp-opt-enable-merging < %s | FileCheck %s
+
+; Merged wrapper should get the max outlined function_entry_count (435).
+
+%struct.ident_t = type { i32, i32, i32, i32, ptr }
+
+ at 0 = private unnamed_addr constant [23 x i8] c";unknown;unknown;0;0;;\00", align 1
+ at 1 = private unnamed_addr constant %struct.ident_t { i32 0, i32 2, i32 0, i32 22, ptr @0 }, align 8
+
+declare void @use(i32 noundef) local_unnamed_addr
+
+define dso_local void @merge_profile(i32 noundef %a, i32 noundef %cond) local_unnamed_addr !prof !13 {
+entry:
+  %a.addr = alloca i32, align 4
+  %cond.addr = alloca i32, align 4
+  store i32 %a, ptr %a.addr, align 4
+  store i32 %cond, ptr %cond.addr, align 4
+  call void (ptr, i32, ptr, ...) @__kmpc_fork_call(ptr nonnull @1, i32 2, ptr nonnull @merge_profile.omp_outlined, ptr nonnull %cond.addr, ptr nonnull %a.addr)
+  call void (ptr, i32, ptr, ...) @__kmpc_fork_call(ptr nonnull @1, i32 2, ptr nonnull @merge_profile.omp_outlined.1, ptr nonnull %cond.addr, ptr nonnull %a.addr)
+  ret void
+}
+
+define internal void @merge_profile.omp_outlined(ptr noalias readnone captures(none) %0, ptr noalias readnone captures(none) %1, ptr noundef nonnull readonly align 4 dereferenceable(4) %cond, ptr noundef nonnull readonly align 4 dereferenceable(4) %a) !prof !14 {
+entry:
+  %v = load i32, ptr %a, align 4
+  tail call void @use(i32 noundef %v)
+  ret void
+}
+
+define internal void @merge_profile.omp_outlined.1(ptr noalias readnone captures(none) %0, ptr noalias readnone captures(none) %1, ptr noundef nonnull readonly align 4 dereferenceable(4) %cond, ptr noundef nonnull readonly align 4 dereferenceable(4) %a) !prof !15 {
+entry:
+  %c = load i32, ptr %cond, align 4
+  %t = icmp eq i32 %c, 0
+  %v = load i32, ptr %a, align 4
+  %off = select i1 %t, i32 3, i32 2, !prof !16
+  %sum = add nsw i32 %v, %off
+  tail call void @use(i32 noundef %sum)
+  ret void
+}
+
+declare !callback !17 void @__kmpc_fork_call(ptr, i32, ptr, ...) local_unnamed_addr
+
+declare i32 @__kmpc_global_thread_num(ptr) local_unnamed_addr
+declare void @__kmpc_barrier(ptr, i32) local_unnamed_addr
+
+!llvm.module.flags = !{!0, !1}
+
+!0 = !{i32 7, !"openmp", i32 51}
+!1 = !{i32 1, !"ProfileSummary", !2}
+!2 = !{!3, !4, !5, !6, !7, !8, !9, !10}
+!3 = !{!"ProfileFormat", !"InstrProf"}
----------------
alokkrsharma wrote:

Thanks, updated accordingly. The test now uses SampleProfile, and I've added a note as well.

https://github.com/llvm/llvm-project/pull/221654


More information about the llvm-commits mailing list