[llvm] [MergeFunctions] Combine instruction metadata instead of comparing it (PR #225921)

Mian Miftah via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 29 20:52:13 PDT 2026


https://github.com/mmiftahx updated https://github.com/llvm/llvm-project/pull/225921

>From 7b26b1e815d7fa227b1051eb4e0e93d141ecf38a Mon Sep 17 00:00:00 2001
From: mmiftahx <mmiftah.duna at gmail.com>
Date: Wed, 23 Sep 2026 02:12:39 -0500
Subject: [PATCH 1/3] [MergeFunctions] Precommit tests for combining metadata
 (NFC)

Regenerate the range and nonnull tests with update_test_checks.py, and
add tests for merging functions that differ in instruction metadata.
---
 .../assignment-tracking/mergefunc/merge.ll    |  49 ++
 .../MergeFunc/call-and-invoke-with-ranges.ll  |  81 ++-
 .../MergeFunc/instruction-metadata.ll         | 647 ++++++++++++++++++
 .../MergeFunc/mergefunc-preserve-nonnull.ll   |  75 +-
 .../Transforms/MergeFunc/ranges-multiple.ll   |  46 +-
 llvm/test/Transforms/MergeFunc/ranges.ll      |  46 +-
 6 files changed, 875 insertions(+), 69 deletions(-)
 create mode 100644 llvm/test/DebugInfo/Generic/assignment-tracking/mergefunc/merge.ll
 create mode 100644 llvm/test/Transforms/MergeFunc/instruction-metadata.ll

diff --git a/llvm/test/DebugInfo/Generic/assignment-tracking/mergefunc/merge.ll b/llvm/test/DebugInfo/Generic/assignment-tracking/mergefunc/merge.ll
new file mode 100644
index 0000000000000..f73fb8aef8aa4
--- /dev/null
+++ b/llvm/test/DebugInfo/Generic/assignment-tracking/mergefunc/merge.ll
@@ -0,0 +1,49 @@
+; RUN: opt -passes=mergefunc -S < %s | FileCheck %s --implicit-check-not='@second'
+
+; Assignment IDs link a store to the debug records of the function that
+; contains it. Keep the IDs of the surviving function.
+
+; CHECK-LABEL: define void @first(
+; CHECK-NEXT: store i8 %v, ptr %p, align 1, !DIAssignID ![[ID:[0-9]+]]
+; CHECK-NEXT: #dbg_assign(i8 %v, ![[VAR:[0-9]+]], !DIExpression(), ![[ID]], ptr %p, !DIExpression(), ![[LOC:[0-9]+]])
+; CHECK-NEXT: ret void
+; CHECK: ![[SP:[0-9]+]] = distinct !DISubprogram(name: "first",
+; CHECK: ![[ID]] = distinct !DIAssignID()
+; CHECK: ![[VAR]] = !DILocalVariable(name: "value", scope: ![[SP]],
+; CHECK: ![[LOC]] = !DILocation(line: 1, scope: ![[SP]])
+define void @first(ptr %p, i8 %v) !dbg !5 {
+  store i8 %v, ptr %p, !DIAssignID !9
+  #dbg_assign(i8 %v, !7, !DIExpression(), !9, ptr %p, !DIExpression(), !11)
+  ret void
+}
+
+define internal void @second(ptr %p, i8 %v) !dbg !6 {
+  store i8 %v, ptr %p, !DIAssignID !10
+  #dbg_assign(i8 %v, !8, !DIExpression(), !10, ptr %p, !DIExpression(), !12)
+  ret void
+}
+
+define void @calls(ptr %p, i8 %v) {
+  call void @first(ptr %p, i8 %v)
+  call void @second(ptr %p, i8 %v)
+  ret void
+}
+
+!llvm.dbg.cu = !{!0}
+!llvm.module.flags = !{!3, !4}
+
+!0 = distinct !DICompileUnit(language: DW_LANG_C, file: !1, producer: "clang", isOptimized: true, runtimeVersion: 0, emissionKind: FullDebug)
+!1 = !DIFile(filename: "merge.c", directory: "/")
+!2 = !DISubroutineType(types: !13)
+!3 = !{i32 2, !"Debug Info Version", i32 3}
+!4 = !{i32 7, !"debug-info-assignment-tracking", i1 true}
+!5 = distinct !DISubprogram(name: "first", scope: !1, file: !1, line: 1, type: !2, scopeLine: 1, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
+!6 = distinct !DISubprogram(name: "second", scope: !1, file: !1, line: 2, type: !2, scopeLine: 2, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !0)
+!7 = !DILocalVariable(name: "value", scope: !5, file: !1, line: 1, type: !14)
+!8 = !DILocalVariable(name: "value", scope: !6, file: !1, line: 2, type: !14)
+!9 = distinct !DIAssignID()
+!10 = distinct !DIAssignID()
+!11 = !DILocation(line: 1, scope: !5)
+!12 = !DILocation(line: 2, scope: !6)
+!13 = !{}
+!14 = !DIBasicType(name: "char", size: 8, encoding: DW_ATE_signed_char)
diff --git a/llvm/test/Transforms/MergeFunc/call-and-invoke-with-ranges.ll b/llvm/test/Transforms/MergeFunc/call-and-invoke-with-ranges.ll
index 39e5a11181a4f..7ca189ceaa9a3 100644
--- a/llvm/test/Transforms/MergeFunc/call-and-invoke-with-ranges.ll
+++ b/llvm/test/Transforms/MergeFunc/call-and-invoke-with-ranges.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --include-generated-funcs --version 6
 ; RUN: opt -passes=mergefunc -S < %s | FileCheck %s
 
 define i8 @call_with_range() {
@@ -7,20 +8,12 @@ define i8 @call_with_range() {
 }
 
 define i8 @call_no_range() {
-; CHECK-LABEL: @call_no_range
-; CHECK-NEXT: bitcast i8 0 to i8
-; CHECK-NEXT: %out = call i8 @dummy()
-; CHECK-NEXT: ret i8 %out
   bitcast i8 0 to i8
   %out = call i8 @dummy()
   ret i8 %out
 }
 
 define i8 @call_different_range() {
-; CHECK-LABEL: @call_different_range
-; CHECK-NEXT: bitcast i8 0 to i8
-; CHECK-NEXT: %out = call i8 @dummy(), !range !1
-; CHECK-NEXT: ret i8 %out
   bitcast i8 0 to i8
   %out = call i8 @dummy(), !range !1
   ret i8 %out
@@ -38,8 +31,6 @@ lpad:
 }
 
 define i8 @invoke_no_range() personality ptr undef {
-; CHECK-LABEL: @invoke_no_range()
-; CHECK-NEXT: invoke i8 @dummy
   %out = invoke i8 @dummy() to label %next unwind label %lpad
 
 next:
@@ -51,8 +42,6 @@ lpad:
 }
 
 define i8 @invoke_different_range() personality ptr undef {
-; CHECK-LABEL: @invoke_different_range()
-; CHECK-NEXT: invoke i8 @dummy
   %out = invoke i8 @dummy() to label %next unwind label %lpad, !range !1
 
 next:
@@ -64,8 +53,6 @@ lpad:
 }
 
 define i8 @invoke_with_same_range() personality ptr undef {
-; CHECK-DAG: @invoke_with_same_range()
-; CHECK-DAG: tail call i8 @invoke_with_range()
   %out = invoke i8 @dummy() to label %next unwind label %lpad, !range !0
 
 next:
@@ -77,8 +64,6 @@ lpad:
 }
 
 define i8 @call_with_same_range() {
-; CHECK-DAG: @call_with_same_range
-; CHECK-DAG: tail call i8 @call_with_range
   bitcast i8 0 to i8
   %out = call i8 @dummy(), !range !0
   ret i8 %out
@@ -90,3 +75,67 @@ declare i32 @__gxx_personality_v0(...)
 
 !0 = !{i8 0, i8 2}
 !1 = !{i8 5, i8 7}
+; CHECK-LABEL: define i8 @call_with_range() {
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i8 0 to i8
+; CHECK-NEXT:    [[OUT:%.*]] = call i8 @dummy(), !range [[RNG0:![0-9]+]]
+; CHECK-NEXT:    ret i8 [[OUT]]
+;
+;
+; CHECK-LABEL: define i8 @call_no_range() {
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i8 0 to i8
+; CHECK-NEXT:    [[OUT:%.*]] = call i8 @dummy()
+; CHECK-NEXT:    ret i8 [[OUT]]
+;
+;
+; CHECK-LABEL: define i8 @call_different_range() {
+; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i8 0 to i8
+; CHECK-NEXT:    [[OUT:%.*]] = call i8 @dummy(), !range [[RNG1:![0-9]+]]
+; CHECK-NEXT:    ret i8 [[OUT]]
+;
+;
+; CHECK-LABEL: define i8 @invoke_with_range() personality ptr undef {
+; CHECK-NEXT:    [[OUT:%.*]] = invoke i8 @dummy()
+; CHECK-NEXT:            to label %[[NEXT:.*]] unwind label %[[LPAD:.*]], !range [[RNG0]]
+; CHECK:       [[NEXT]]:
+; CHECK-NEXT:    ret i8 [[OUT]]
+; CHECK:       [[LPAD]]:
+; CHECK-NEXT:    [[PAD:%.*]] = landingpad { ptr, i32 }
+; CHECK-NEXT:            cleanup
+; CHECK-NEXT:    resume { ptr, i32 } zeroinitializer
+;
+;
+; CHECK-LABEL: define i8 @invoke_no_range() personality ptr undef {
+; CHECK-NEXT:    [[OUT:%.*]] = invoke i8 @dummy()
+; CHECK-NEXT:            to label %[[NEXT:.*]] unwind label %[[LPAD:.*]]
+; CHECK:       [[NEXT]]:
+; CHECK-NEXT:    ret i8 [[OUT]]
+; CHECK:       [[LPAD]]:
+; CHECK-NEXT:    [[PAD:%.*]] = landingpad { ptr, i32 }
+; CHECK-NEXT:            cleanup
+; CHECK-NEXT:    resume { ptr, i32 } zeroinitializer
+;
+;
+; CHECK-LABEL: define i8 @invoke_different_range() personality ptr undef {
+; CHECK-NEXT:    [[OUT:%.*]] = invoke i8 @dummy()
+; CHECK-NEXT:            to label %[[NEXT:.*]] unwind label %[[LPAD:.*]], !range [[RNG1]]
+; CHECK:       [[NEXT]]:
+; CHECK-NEXT:    ret i8 [[OUT]]
+; CHECK:       [[LPAD]]:
+; CHECK-NEXT:    [[PAD:%.*]] = landingpad { ptr, i32 }
+; CHECK-NEXT:            cleanup
+; CHECK-NEXT:    resume { ptr, i32 } zeroinitializer
+;
+;
+; CHECK-LABEL: define i8 @call_with_same_range() {
+; CHECK-NEXT:    [[TMP1:%.*]] = tail call i8 @call_with_range()
+; CHECK-NEXT:    ret i8 [[TMP1]]
+;
+;
+; CHECK-LABEL: define i8 @invoke_with_same_range() personality ptr undef {
+; CHECK-NEXT:    [[TMP1:%.*]] = tail call i8 @invoke_with_range()
+; CHECK-NEXT:    ret i8 [[TMP1]]
+;
+;.
+; CHECK: [[RNG0]] = !{i8 0, i8 2}
+; CHECK: [[RNG1]] = !{i8 5, i8 7}
+;.
diff --git a/llvm/test/Transforms/MergeFunc/instruction-metadata.ll b/llvm/test/Transforms/MergeFunc/instruction-metadata.ll
new file mode 100644
index 0000000000000..327e8901f13f9
--- /dev/null
+++ b/llvm/test/Transforms/MergeFunc/instruction-metadata.ll
@@ -0,0 +1,647 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --version 6
+; RUN: opt -S -passes=mergefunc < %s | FileCheck %s
+
+declare i8 @f()
+declare void @callee()
+declare ptr @malloc(i64)
+declare void @llvm.memcpy.p0.p0.i64(ptr, ptr, i64, i1)
+declare void @llvm.experimental.noalias.scope.decl(metadata)
+declare void @use(ptr)
+
+define i8 @fn_range(ptr %p) {
+; CHECK-LABEL: define i8 @fn_range(
+; CHECK-SAME: ptr [[P:%.*]]) {
+; CHECK-NEXT:    [[V:%.*]] = load i8, ptr [[P]], align 1, !range [[RNG3:![0-9]+]], !noundef [[META4:![0-9]+]]
+; CHECK-NEXT:    [[C:%.*]] = call i8 @f(), !range [[RNG3]]
+; CHECK-NEXT:    [[R:%.*]] = add i8 [[V]], [[C]]
+; CHECK-NEXT:    ret i8 [[R]]
+;
+  %v = load i8, ptr %p, !range !0, !noundef !2
+  %c = call i8 @f(), !range !0
+  %r = add i8 %v, %c
+  ret i8 %r
+}
+
+define internal i8 @fn_range_b(ptr %p) {
+; CHECK-LABEL: define internal i8 @fn_range_b(
+; CHECK-SAME: ptr [[P:%.*]]) {
+; CHECK-NEXT:    [[V:%.*]] = load i8, ptr [[P]], align 1, !range [[RNG5:![0-9]+]], !noundef [[META4]]
+; CHECK-NEXT:    [[C:%.*]] = call i8 @f(), !range [[RNG5]]
+; CHECK-NEXT:    [[R:%.*]] = add i8 [[V]], [[C]]
+; CHECK-NEXT:    ret i8 [[R]]
+;
+  %v = load i8, ptr %p, !range !1, !noundef !2
+  %c = call i8 @f(), !range !1
+  %r = add i8 %v, %c
+  ret i8 %r
+}
+
+define ptr @fn_nonnull(ptr %p) {
+; CHECK-LABEL: define ptr @fn_nonnull(
+; CHECK-SAME: ptr [[P:%.*]]) {
+; CHECK-NEXT:    [[V:%.*]] = load ptr, ptr [[P]], align 8, !nonnull [[META4]]
+; CHECK-NEXT:    ret ptr [[V]]
+;
+  %v = load ptr, ptr %p, !nonnull !2
+  ret ptr %v
+}
+
+define internal ptr @fn_nonnull_b(ptr %p) {
+; CHECK-LABEL: define internal ptr @fn_nonnull_b(
+; CHECK-SAME: ptr [[P:%.*]]) {
+; CHECK-NEXT:    [[V:%.*]] = load ptr, ptr [[P]], align 8
+; CHECK-NEXT:    ret ptr [[V]]
+;
+  %v = load ptr, ptr %p
+  ret ptr %v
+}
+
+define void @fn_tbaa(ptr %p) {
+; CHECK-LABEL: define void @fn_tbaa(
+; CHECK-SAME: ptr [[P:%.*]]) {
+; CHECK-NEXT:    store i32 0, ptr [[P]], align 4, !tbaa [[INT_TBAA6:![0-9]+]]
+; CHECK-NEXT:    ret void
+;
+  store i32 0, ptr %p, !tbaa !3
+  ret void
+}
+
+define internal void @fn_tbaa_b(ptr %p) {
+  store i32 0, ptr %p, !tbaa !7
+  ret void
+}
+
+define void @fn_callees(ptr %fp) {
+; CHECK-LABEL: define void @fn_callees(
+; CHECK-SAME: ptr [[FP:%.*]]) {
+; CHECK-NEXT:    call void [[FP]](), !callees [[META10:![0-9]+]]
+; CHECK-NEXT:    ret void
+;
+  call void %fp(), !callees !9
+  ret void
+}
+
+define internal void @fn_callees_b(ptr %fp) {
+  call void %fp()
+  ret void
+}
+
+; AMDGPU atomic metadata and !atomic.ignore.denormal.mode are kept if they are
+; the same in both functions. Other unknown metadata is dropped.
+define float @fn_atomic(ptr %p, float %v) {
+; CHECK-LABEL: define float @fn_atomic(
+; CHECK-SAME: ptr [[P:%.*]], float [[V:%.*]]) {
+; CHECK-NEXT:    [[R:%.*]] = atomicrmw fadd ptr [[P]], float [[V]] monotonic, align 4, !atomic.ignore.denormal.mode [[META4]], !amdgpu.no.fine.grained.memory [[META4]], !amdgpu.no.remote.memory [[META4]], !custom.kind [[META4]]
+; CHECK-NEXT:    [[S:%.*]] = atomicrmw fadd ptr [[P]], float [[R]] monotonic, align 4, !amdgpu.no.fine.grained.memory [[META4]]
+; CHECK-NEXT:    ret float [[S]]
+;
+  %r = atomicrmw fadd ptr %p, float %v monotonic, !amdgpu.no.fine.grained.memory !2, !amdgpu.no.remote.memory !2, !atomic.ignore.denormal.mode !2, !custom.kind !2
+  %s = atomicrmw fadd ptr %p, float %r monotonic, !amdgpu.no.fine.grained.memory !2
+  ret float %s
+}
+
+define internal float @fn_atomic_b(ptr %p, float %v) {
+  %r = atomicrmw fadd ptr %p, float %v monotonic, !amdgpu.no.remote.memory !2, !atomic.ignore.denormal.mode !2, !custom.kind !2
+  %s = atomicrmw fadd ptr %p, float %r monotonic, !amdgpu.no.fine.grained.memory !2
+  ret float %s
+}
+
+; !tbaa.struct and !srcloc are kept if they are the same in both functions.
+define void @fn_same(ptr %p, ptr %q) {
+; CHECK-LABEL: define void @fn_same(
+; CHECK-SAME: ptr [[P:%.*]], ptr [[Q:%.*]]) {
+; CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i64(ptr [[P]], ptr [[Q]], i64 4, i1 false), !tbaa.struct [[TBAA_STRUCT11:![0-9]+]]
+; CHECK-NEXT:    call void asm sideeffect "", ""(), !srcloc [[META12:![0-9]+]]
+; CHECK-NEXT:    ret void
+;
+  call void @llvm.memcpy.p0.p0.i64(ptr %p, ptr %q, i64 4, i1 false), !tbaa.struct !35
+  call void asm sideeffect "", ""(), !srcloc !36
+  ret void
+}
+
+define internal void @fn_same_b(ptr %p, ptr %q) {
+  call void @llvm.memcpy.p0.p0.i64(ptr %p, ptr %q, i64 4, i1 false), !tbaa.struct !35
+  call void asm sideeffect "", ""(), !srcloc !36
+  ret void
+}
+
+; A noalias scope declaration sets how long its scope lasts. !42 lasts for the
+; whole call of @fn_scope_decl, whose loop declares the unrelated scope !43, but
+; only for one iteration of the loop of @fn_scope_decl_b. The same attachments
+; say different things, so they are dropped.
+define void @fn_scope_decl(ptr %p, ptr %q, i64 %n) {
+; CHECK-LABEL: define void @fn_scope_decl(
+; CHECK-SAME: ptr [[P:%.*]], ptr [[Q:%.*]], i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    call void @llvm.experimental.noalias.scope.decl(metadata [[META13:![0-9]+]])
+; CHECK-NEXT:    [[PI:%.*]] = getelementptr i32, ptr [[P]], i64 [[I]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[PI]], align 4, !alias.scope [[META16:![0-9]+]]
+; CHECK-NEXT:    [[QI:%.*]] = getelementptr i32, ptr [[Q]], i64 [[I]]
+; CHECK-NEXT:    store i32 [[V]], ptr [[QI]], align 4, !noalias [[META16]]
+; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  call void @llvm.experimental.noalias.scope.decl(metadata !45)
+  %pi = getelementptr i32, ptr %p, i64 %i
+  %v = load i32, ptr %pi, !alias.scope !44
+  %qi = getelementptr i32, ptr %q, i64 %i
+  store i32 %v, ptr %qi, !noalias !44
+  %i.next = add i64 %i, 1
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+define internal void @fn_scope_decl_b(ptr %p, ptr %q, i64 %n) {
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  call void @llvm.experimental.noalias.scope.decl(metadata !44)
+  %pi = getelementptr i32, ptr %p, i64 %i
+  %v = load i32, ptr %pi, !alias.scope !44
+  %qi = getelementptr i32, ptr %q, i64 %i
+  store i32 %v, ptr %qi, !noalias !44
+  %i.next = add i64 %i, 1
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+; TODO: The scopes correspond one to one, so they could be kept.
+define i32 @fn_scopes(ptr %a, ptr %b) {
+; CHECK-LABEL: define i32 @fn_scopes(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) {
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[B]], align 4, !noalias [[META18:![0-9]+]]
+; CHECK-NEXT:    store i32 0, ptr [[A]], align 4, !alias.scope [[META18]]
+; CHECK-NEXT:    ret i32 [[V]]
+;
+  %v = load i32, ptr %b, !noalias !10
+  store i32 0, ptr %a, !alias.scope !10
+  ret i32 %v
+}
+
+define internal i32 @fn_scopes_b(ptr %a, ptr %b) {
+  %v = load i32, ptr %b, !noalias !13
+  store i32 0, ptr %a, !alias.scope !13
+  ret i32 %v
+}
+
+; Loop metadata is kept only if both loops have the same properties. The
+; debug locations of the loops may differ.
+define void @fn_loop(i64 %n) !dbg !29 {
+; CHECK-LABEL: define void @fn_loop(
+; CHECK-SAME: i64 [[N:%.*]]) !dbg [[DBG21:![0-9]+]] {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 1
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP23:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  %i.next = add i64 %i, 1
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop, !llvm.loop !16
+
+exit:
+  ret void
+}
+
+define internal void @fn_loop_b(i64 %n) !dbg !32 {
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  %i.next = add i64 %i, 1
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop, !llvm.loop !18
+
+exit:
+  ret void
+}
+
+define void @fn_loop_diff(i64 %n) {
+; CHECK-LABEL: define void @fn_loop_diff(
+; CHECK-SAME: i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 2
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP27:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  %i.next = add i64 %i, 2
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop, !llvm.loop !19
+
+exit:
+  ret void
+}
+
+define internal void @fn_loop_diff_b(i64 %n) {
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  %i.next = add i64 %i, 2
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop
+
+exit:
+  ret void
+}
+
+; The verifier accepts !llvm.loop without operands, although it is not a valid
+; loop ID.
+define void @fn_loop_empty(i64 %n) {
+; CHECK-LABEL: define void @fn_loop_empty(
+; CHECK-SAME: i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 3
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[META4]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  %i.next = add i64 %i, 3
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop, !llvm.loop !2
+
+exit:
+  ret void
+}
+
+define internal void @fn_loop_empty_b(i64 %n) {
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  %i.next = add i64 %i, 3
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop, !llvm.loop !2
+
+exit:
+  ret void
+}
+
+; Loop::getLoopID() ignores a node whose first operand is not the node itself,
+; so the loop of @fn_loop_invalid_b has no properties, and the loop metadata is
+; dropped.
+define void @fn_loop_invalid(i64 %n) {
+; CHECK-LABEL: define void @fn_loop_invalid(
+; CHECK-SAME: i64 [[N:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[LOOP:.*]]
+; CHECK:       [[LOOP]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
+; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 4
+; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP28:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  %i.next = add i64 %i, 4
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop, !llvm.loop !46
+
+exit:
+  ret void
+}
+
+define internal void @fn_loop_invalid_b(i64 %n) {
+entry:
+  br label %loop
+
+loop:
+  %i = phi i64 [ 0, %entry ], [ %i.next, %loop ]
+  %i.next = add i64 %i, 4
+  %done = icmp eq i64 %i.next, %n
+  br i1 %done, label %exit, label %loop, !llvm.loop !47
+
+exit:
+  ret void
+}
+
+; !llvm.mem.parallel_loop_access names loop IDs. !38 is the inner loop of
+; @fn_parallel but the outer loop of @fn_parallel_b, so the same attachment
+; says that different loops are parallel. It is dropped.
+define void @fn_parallel(ptr %a, ptr %b) {
+; CHECK-LABEL: define void @fn_parallel(
+; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) {
+; CHECK-NEXT:  [[ENTRY:.*]]:
+; CHECK-NEXT:    br label %[[OUTER:.*]]
+; CHECK:       [[OUTER]]:
+; CHECK-NEXT:    [[O:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[O_NEXT:%.*]], %[[LATCH:.*]] ]
+; CHECK-NEXT:    br label %[[INNER:.*]]
+; CHECK:       [[INNER]]:
+; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[OUTER]] ], [ [[I_NEXT:%.*]], %[[INNER]] ]
+; CHECK-NEXT:    [[IDX:%.*]] = add i64 [[O]], [[I]]
+; CHECK-NEXT:    [[PA:%.*]] = getelementptr i32, ptr [[A]], i64 [[IDX]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[PA]], align 4, !llvm.mem.parallel_loop_access [[META29:![0-9]+]]
+; CHECK-NEXT:    [[PB:%.*]] = getelementptr i32, ptr [[B]], i64 [[IDX]]
+; CHECK-NEXT:    store i32 [[V]], ptr [[PB]], align 4, !llvm.mem.parallel_loop_access [[META29]]
+; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 1
+; CHECK-NEXT:    [[I_DONE:%.*]] = icmp eq i64 [[I_NEXT]], 16
+; CHECK-NEXT:    br i1 [[I_DONE]], label %[[LATCH]], label %[[INNER]], !llvm.loop [[LOOP30:![0-9]+]]
+; CHECK:       [[LATCH]]:
+; CHECK-NEXT:    [[O_NEXT]] = add i64 [[O]], 16
+; CHECK-NEXT:    [[O_DONE:%.*]] = icmp eq i64 [[O_NEXT]], 64
+; CHECK-NEXT:    br i1 [[O_DONE]], label %[[EXIT:.*]], label %[[OUTER]], !llvm.loop [[LOOP31:![0-9]+]]
+; CHECK:       [[EXIT]]:
+; CHECK-NEXT:    ret void
+;
+entry:
+  br label %outer
+
+outer:
+  %o = phi i64 [ 0, %entry ], [ %o.next, %latch ]
+  br label %inner
+
+inner:
+  %i = phi i64 [ 0, %outer ], [ %i.next, %inner ]
+  %idx = add i64 %o, %i
+  %pa = getelementptr i32, ptr %a, i64 %idx
+  %v = load i32, ptr %pa, !llvm.mem.parallel_loop_access !37
+  %pb = getelementptr i32, ptr %b, i64 %idx
+  store i32 %v, ptr %pb, !llvm.mem.parallel_loop_access !37
+  %i.next = add i64 %i, 1
+  %i.done = icmp eq i64 %i.next, 16
+  br i1 %i.done, label %latch, label %inner, !llvm.loop !38
+
+latch:
+  %o.next = add i64 %o, 16
+  %o.done = icmp eq i64 %o.next, 64
+  br i1 %o.done, label %exit, label %outer, !llvm.loop !39
+
+exit:
+  ret void
+}
+
+define internal void @fn_parallel_b(ptr %a, ptr %b) {
+entry:
+  br label %outer
+
+outer:
+  %o = phi i64 [ 0, %entry ], [ %o.next, %latch ]
+  br label %inner
+
+inner:
+  %i = phi i64 [ 0, %outer ], [ %i.next, %inner ]
+  %idx = add i64 %o, %i
+  %pa = getelementptr i32, ptr %a, i64 %idx
+  %v = load i32, ptr %pa, !llvm.mem.parallel_loop_access !37
+  %pb = getelementptr i32, ptr %b, i64 %idx
+  store i32 %v, ptr %pb, !llvm.mem.parallel_loop_access !37
+  %i.next = add i64 %i, 1
+  %i.done = icmp eq i64 %i.next, 16
+  br i1 %i.done, label %latch, label %inner, !llvm.loop !40
+
+latch:
+  %o.next = add i64 %o, 16
+  %o.done = icmp eq i64 %o.next, 64
+  br i1 %o.done, label %exit, label %outer, !llvm.loop !38
+
+exit:
+  ret void
+}
+
+; Without !coro.outside.frame the alloca could be moved into the coroutine
+; frame, so the attachment of the kept function stays.
+define void @fn_coro() {
+; CHECK-LABEL: define void @fn_coro() {
+; CHECK-NEXT:    [[A:%.*]] = alloca i32, align 4, !coro.outside.frame [[META4]]
+; CHECK-NEXT:    call void @use(ptr [[A]])
+; CHECK-NEXT:    ret void
+;
+  %a = alloca i32, align 4, !coro.outside.frame !2
+  call void @use(ptr %a)
+  ret void
+}
+
+define internal void @fn_coro_b() {
+  %a = alloca i32, align 4, !coro.outside.frame !2
+  call void @use(ptr %a)
+  ret void
+}
+
+; Keep the memprof metadata of the surviving function.
+define ptr @fn_memprof(i64 %n) {
+; CHECK-LABEL: define ptr @fn_memprof(
+; CHECK-SAME: i64 [[N:%.*]]) {
+; CHECK-NEXT:    [[P:%.*]] = call ptr @malloc(i64 [[N]]), !callsite [[META32:![0-9]+]]
+; CHECK-NEXT:    ret ptr [[P]]
+;
+  %p = call ptr @malloc(i64 %n), !callsite !20
+  ret ptr %p
+}
+
+define internal ptr @fn_memprof_b(i64 %n) {
+  %p = call ptr @malloc(i64 %n), !memprof !21, !callsite !24
+  ret ptr %p
+}
+
+define void @calls(ptr %p, i64 %n, float %v) {
+; CHECK-LABEL: define void @calls(
+; CHECK-SAME: ptr [[P:%.*]], i64 [[N:%.*]], float [[V:%.*]]) {
+; CHECK-NEXT:    [[TMP1:%.*]] = call i8 @fn_range(ptr [[P]])
+; CHECK-NEXT:    [[TMP2:%.*]] = call i8 @fn_range_b(ptr [[P]])
+; CHECK-NEXT:    [[TMP3:%.*]] = call ptr @fn_nonnull(ptr [[P]])
+; CHECK-NEXT:    [[TMP4:%.*]] = call ptr @fn_nonnull_b(ptr [[P]])
+; CHECK-NEXT:    call void @fn_tbaa(ptr [[P]])
+; CHECK-NEXT:    call void @fn_tbaa(ptr [[P]])
+; CHECK-NEXT:    call void @fn_callees(ptr [[P]])
+; CHECK-NEXT:    call void @fn_callees(ptr [[P]])
+; CHECK-NEXT:    [[TMP5:%.*]] = call float @fn_atomic(ptr [[P]], float [[V]])
+; CHECK-NEXT:    [[TMP6:%.*]] = call float @fn_atomic(ptr [[P]], float [[V]])
+; CHECK-NEXT:    call void @fn_same(ptr [[P]], ptr [[P]])
+; CHECK-NEXT:    call void @fn_same(ptr [[P]], ptr [[P]])
+; CHECK-NEXT:    [[TMP7:%.*]] = call i32 @fn_scopes(ptr [[P]], ptr [[P]])
+; CHECK-NEXT:    [[TMP8:%.*]] = call i32 @fn_scopes(ptr [[P]], ptr [[P]])
+; CHECK-NEXT:    call void @fn_scope_decl(ptr [[P]], ptr [[P]], i64 [[N]])
+; CHECK-NEXT:    call void @fn_scope_decl(ptr [[P]], ptr [[P]], i64 [[N]])
+; CHECK-NEXT:    call void @fn_loop(i64 [[N]])
+; CHECK-NEXT:    call void @fn_loop(i64 [[N]])
+; CHECK-NEXT:    call void @fn_loop_diff(i64 [[N]])
+; CHECK-NEXT:    call void @fn_loop_diff(i64 [[N]])
+; CHECK-NEXT:    call void @fn_loop_empty(i64 [[N]])
+; CHECK-NEXT:    call void @fn_loop_empty(i64 [[N]])
+; CHECK-NEXT:    call void @fn_loop_invalid(i64 [[N]])
+; CHECK-NEXT:    call void @fn_loop_invalid(i64 [[N]])
+; CHECK-NEXT:    call void @fn_parallel(ptr [[P]], ptr [[P]])
+; CHECK-NEXT:    call void @fn_parallel(ptr [[P]], ptr [[P]])
+; CHECK-NEXT:    call void @fn_coro()
+; CHECK-NEXT:    call void @fn_coro()
+; CHECK-NEXT:    [[TMP9:%.*]] = call ptr @fn_memprof(i64 [[N]])
+; CHECK-NEXT:    [[TMP10:%.*]] = call ptr @fn_memprof(i64 [[N]])
+; CHECK-NEXT:    ret void
+;
+  call i8 @fn_range(ptr %p)
+  call i8 @fn_range_b(ptr %p)
+  call ptr @fn_nonnull(ptr %p)
+  call ptr @fn_nonnull_b(ptr %p)
+  call void @fn_tbaa(ptr %p)
+  call void @fn_tbaa_b(ptr %p)
+  call void @fn_callees(ptr %p)
+  call void @fn_callees_b(ptr %p)
+  call float @fn_atomic(ptr %p, float %v)
+  call float @fn_atomic_b(ptr %p, float %v)
+  call void @fn_same(ptr %p, ptr %p)
+  call void @fn_same_b(ptr %p, ptr %p)
+  call i32 @fn_scopes(ptr %p, ptr %p)
+  call i32 @fn_scopes_b(ptr %p, ptr %p)
+  call void @fn_scope_decl(ptr %p, ptr %p, i64 %n)
+  call void @fn_scope_decl_b(ptr %p, ptr %p, i64 %n)
+  call void @fn_loop(i64 %n)
+  call void @fn_loop_b(i64 %n)
+  call void @fn_loop_diff(i64 %n)
+  call void @fn_loop_diff_b(i64 %n)
+  call void @fn_loop_empty(i64 %n)
+  call void @fn_loop_empty_b(i64 %n)
+  call void @fn_loop_invalid(i64 %n)
+  call void @fn_loop_invalid_b(i64 %n)
+  call void @fn_parallel(ptr %p, ptr %p)
+  call void @fn_parallel_b(ptr %p, ptr %p)
+  call void @fn_coro()
+  call void @fn_coro_b()
+  call ptr @fn_memprof(i64 %n)
+  call ptr @fn_memprof_b(i64 %n)
+  ret void
+}
+
+!llvm.dbg.cu = !{!25}
+!llvm.module.flags = !{!27}
+
+!0 = !{i8 0, i8 2}
+!1 = !{i8 5, i8 7}
+!2 = !{}
+!3 = !{!4, !4, i64 0}
+!4 = !{!"int", !5, i64 0}
+!5 = !{!"omnipotent char", !6, i64 0}
+!6 = !{!"Simple C/C++ TBAA"}
+!7 = !{!8, !8, i64 0}
+!8 = !{!"float", !5, i64 0}
+!9 = !{ptr @callee}
+!10 = !{!11}
+!11 = distinct !{!11, !12, !"scope"}
+!12 = distinct !{!12, !"domain"}
+!13 = !{!14}
+!14 = distinct !{!14, !15, !"scope"}
+!15 = distinct !{!15, !"domain"}
+!16 = distinct !{!16, !30, !31, !17}
+!17 = !{!"llvm.loop.mustprogress"}
+!18 = distinct !{!18, !33, !34, !17}
+!19 = distinct !{!19, !17}
+!20 = !{i64 111}
+!21 = !{!22}
+!22 = !{!23, !"cold"}
+!23 = !{i64 222, i64 333}
+!24 = !{i64 222}
+!25 = distinct !DICompileUnit(language: DW_LANG_C11, file: !26, producer: "clang", isOptimized: true, runtimeVersion: 0, emissionKind: FullDebug)
+!26 = !DIFile(filename: "loop.c", directory: "")
+!27 = !{i32 2, !"Debug Info Version", i32 3}
+!28 = !DISubroutineType(types: !2)
+!29 = distinct !DISubprogram(name: "fn_loop", scope: !26, file: !26, line: 1, type: !28, scopeLine: 1, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !25)
+!30 = !DILocation(line: 2, column: 3, scope: !29)
+!31 = !DILocation(line: 3, column: 3, scope: !29)
+!32 = distinct !DISubprogram(name: "fn_loop_b", scope: !26, file: !26, line: 5, type: !28, scopeLine: 5, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: !25)
+!33 = !DILocation(line: 6, column: 3, scope: !32)
+!34 = !DILocation(line: 7, column: 3, scope: !32)
+!35 = !{i64 0, i64 4, !3}
+!36 = !{i64 42}
+!37 = !{!38}
+!38 = distinct !{!38}
+!39 = distinct !{!39}
+!40 = distinct !{!40}
+!41 = distinct !{!41, !"domain"}
+!42 = distinct !{!42, !41, !"scope"}
+!43 = distinct !{!43, !41, !"other scope"}
+!44 = !{!42}
+!45 = !{!43}
+!46 = distinct !{!46, !17}
+!47 = distinct !{!2, !17}
+;.
+; CHECK: attributes #[[ATTR0:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(argmem: readwrite) }
+; CHECK: attributes #[[ATTR1:[0-9]+]] = { nocallback nofree nosync nounwind willreturn memory(inaccessiblemem: readwrite) }
+;.
+; CHECK: [[META0:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C11, file: [[META1:![0-9]+]], producer: "clang", isOptimized: true, runtimeVersion: 0, emissionKind: FullDebug)
+; CHECK: [[META1]] = !DIFile(filename: "{{.*}}loop.c", directory: {{.*}})
+; CHECK: [[META2:![0-9]+]] = !{i32 2, !"Debug Info Version", i32 3}
+; CHECK: [[RNG3]] = !{i8 0, i8 2}
+; CHECK: [[META4]] = !{}
+; CHECK: [[RNG5]] = !{i8 5, i8 7}
+; CHECK: [[INT_TBAA6]] = !{[[META7:![0-9]+]], [[META7]], i64 0}
+; CHECK: [[META7]] = !{!"int", [[META8:![0-9]+]], i64 0}
+; CHECK: [[META8]] = !{!"omnipotent char", [[META9:![0-9]+]], i64 0}
+; CHECK: [[META9]] = !{!"Simple C/C++ TBAA"}
+; CHECK: [[META10]] = !{ptr @callee}
+; CHECK: [[TBAA_STRUCT11]] = !{i64 0, i64 4, [[INT_TBAA6]]}
+; CHECK: [[META12]] = !{i64 42}
+; CHECK: [[META13]] = !{[[META14:![0-9]+]]}
+; CHECK: [[META14]] = distinct !{[[META14]], [[META15:![0-9]+]], !"other scope"}
+; CHECK: [[META15]] = distinct !{[[META15]], !"domain"}
+; CHECK: [[META16]] = !{[[META17:![0-9]+]]}
+; CHECK: [[META17]] = distinct !{[[META17]], [[META15]], !"scope"}
+; CHECK: [[META18]] = !{[[META19:![0-9]+]]}
+; CHECK: [[META19]] = distinct !{[[META19]], [[META20:![0-9]+]], !"scope"}
+; CHECK: [[META20]] = distinct !{[[META20]], !"domain"}
+; CHECK: [[DBG21]] = distinct !DISubprogram(name: "fn_loop", scope: [[META1]], file: [[META1]], line: 1, type: [[META22:![0-9]+]], scopeLine: 1, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META0]])
+; CHECK: [[META22]] = !DISubroutineType(types: [[META4]])
+; CHECK: [[LOOP23]] = distinct !{[[LOOP23]], [[META24:![0-9]+]], [[META25:![0-9]+]], [[META26:![0-9]+]]}
+; CHECK: [[META24]] = !DILocation(line: 2, column: 3, scope: [[DBG21]])
+; CHECK: [[META25]] = !DILocation(line: 3, column: 3, scope: [[DBG21]])
+; CHECK: [[META26]] = !{!"llvm.loop.mustprogress"}
+; CHECK: [[LOOP27]] = distinct !{[[LOOP27]], [[META26]]}
+; CHECK: [[LOOP28]] = distinct !{[[LOOP28]], [[META26]]}
+; CHECK: [[META29]] = !{[[LOOP30]]}
+; CHECK: [[LOOP30]] = distinct !{[[LOOP30]]}
+; CHECK: [[LOOP31]] = distinct !{[[LOOP31]]}
+; CHECK: [[META32]] = !{i64 111}
+;.
diff --git a/llvm/test/Transforms/MergeFunc/mergefunc-preserve-nonnull.ll b/llvm/test/Transforms/MergeFunc/mergefunc-preserve-nonnull.ll
index 544bba4909ea1..329378dc705c0 100644
--- a/llvm/test/Transforms/MergeFunc/mergefunc-preserve-nonnull.ll
+++ b/llvm/test/Transforms/MergeFunc/mergefunc-preserve-nonnull.ll
@@ -1,4 +1,4 @@
-; NOTE: Assertions have been autogenerated by utils/update_test_checks.py
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --include-generated-funcs --version 6
 ; RUN: opt -passes=mergefunc -S < %s | FileCheck %s
 
 ; This test makes sure that the mergefunc pass does not merge functions
@@ -7,54 +7,30 @@
 %1 = type ptr
 
 define void @f1(ptr %0, ptr %1) {
-; CHECK-LABEL: @f1(
-; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1:%.*]], align 8, !nonnull !0
-; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0:%.*]], align 8
-; CHECK-NEXT:    ret void
-;
   %3 = load ptr, ptr %1, align 8, !nonnull !0
   store ptr %3, ptr %0, align 8
   ret void
 }
 
 define void @f2(ptr %0, ptr %1) {
-; CHECK-LABEL: @f2(
-; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1:%.*]], align 8
-; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0:%.*]], align 8
-; CHECK-NEXT:    ret void
-;
   %3 = load ptr, ptr %1, align 8
   store ptr %3, ptr %0, align 8
   ret void
 }
 
 define void @noundef(ptr %0, ptr %1) {
-; CHECK-LABEL: @noundef(
-; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1:%.*]], align 8, !noundef !0
-; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0:%.*]], align 8
-; CHECK-NEXT:    ret void
-;
   %3 = load ptr, ptr %1, align 8, !noundef !0
   store ptr %3, ptr %0, align 8
   ret void
 }
 
 define void @noalias_1(ptr %0, ptr %1) {
-; CHECK-LABEL: @noalias_1(
-; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1:%.*]], align 8, !noalias !1
-; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0:%.*]], align 8, !alias.scope !1
-; CHECK-NEXT:    ret void
-;
   %3 = load ptr, ptr %1, align 8, !noalias !4
   store ptr %3, ptr %0, align 8, !alias.scope !4
   ret void
 }
 
 define void @noundef_dbg(ptr %0, ptr %1) {
-; CHECK-LABEL: @noundef_dbg(
-; CHECK-NEXT:    tail call void @noundef(ptr [[TMP0:%.*]], ptr [[TMP1:%.*]])
-; CHECK-NEXT:    ret void
-;
   %3 = load ptr, ptr %1, align 8, !noundef !0, !dbg !8
   store ptr %3, ptr %0, align 8
   ret void
@@ -62,10 +38,6 @@ define void @noundef_dbg(ptr %0, ptr %1) {
 
 ; FIXME: This is merged despite different noalias metadata.
 define void @noalias_2(ptr %0, ptr %1) {
-; CHECK-LABEL: @noalias_2(
-; CHECK-NEXT:    tail call void @noalias_1(ptr [[TMP0:%.*]], ptr [[TMP1:%.*]])
-; CHECK-NEXT:    ret void
-;
   %3 = load ptr, ptr %1, align 8, !noalias !7
   store ptr %3, ptr %0, align 8, !alias.scope !7
   ret void
@@ -80,3 +52,48 @@ define void @noalias_2(ptr %0, ptr %1) {
 !6 = !{!6, !5}
 !7 = !{!6}
 !8 = !DILocation(line: 1, column: 1, scope: !{})
+; CHECK-LABEL: define void @f1(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1]], align 8, !nonnull [[META0:![0-9]+]]
+; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0]], align 8
+; CHECK-NEXT:    ret void
+;
+;
+; CHECK-LABEL: define void @f2(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1]], align 8
+; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0]], align 8
+; CHECK-NEXT:    ret void
+;
+;
+; CHECK-LABEL: define void @noundef(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1]], align 8, !noundef [[META0]]
+; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0]], align 8
+; CHECK-NEXT:    ret void
+;
+;
+; CHECK-LABEL: define void @noalias_1(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1]], align 8, !noalias [[META1:![0-9]+]]
+; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0]], align 8, !alias.scope [[META1]]
+; CHECK-NEXT:    ret void
+;
+;
+; CHECK-LABEL: define void @noundef_dbg(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    tail call void @noundef(ptr [[TMP0]], ptr [[TMP1]])
+; CHECK-NEXT:    ret void
+;
+;
+; CHECK-LABEL: define void @noalias_2(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    tail call void @noalias_1(ptr [[TMP0]], ptr [[TMP1]])
+; CHECK-NEXT:    ret void
+;
+;.
+; CHECK: [[META0]] = !{}
+; CHECK: [[META1]] = !{[[META2:![0-9]+]]}
+; CHECK: [[META2]] = distinct !{[[META2]], [[META3:![0-9]+]]}
+; CHECK: [[META3]] = distinct !{[[META3]]}
+;.
diff --git a/llvm/test/Transforms/MergeFunc/ranges-multiple.ll b/llvm/test/Transforms/MergeFunc/ranges-multiple.ll
index 181c084e9cf9c..198a457361208 100644
--- a/llvm/test/Transforms/MergeFunc/ranges-multiple.ll
+++ b/llvm/test/Transforms/MergeFunc/ranges-multiple.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --include-generated-funcs --version 6
 ; RUN: opt -passes=mergefunc -S < %s | FileCheck %s
 define i1 @cmp_with_range(ptr, ptr) {
   %v1 = load i8, ptr %0, !range !0
@@ -7,11 +8,6 @@ define i1 @cmp_with_range(ptr, ptr) {
 }
 
 define i1 @cmp_no_range(ptr, ptr) {
-; CHECK-LABEL: @cmp_no_range
-; CHECK-NEXT: %v1 = load i8, ptr %0
-; CHECK-NEXT:  %v2 = load i8, ptr %1
-; CHECK-NEXT:  %out = icmp eq i8 %v1, %v2
-; CHECK-NEXT:  ret i1 %out
   %v1 = load i8, ptr %0
   %v2 = load i8, ptr %1
   %out = icmp eq i8 %v1, %v2
@@ -19,11 +15,6 @@ define i1 @cmp_no_range(ptr, ptr) {
 }
 
 define i1 @cmp_different_range(ptr, ptr) {
-; CHECK-LABEL: @cmp_different_range
-; CHECK-NEXT:  %v1 = load i8, ptr %0, align 1, !range !1
-; CHECK-NEXT:  %v2 = load i8, ptr %1, align 1, !range !1
-; CHECK-NEXT:  %out = icmp eq i8 %v1, %v2
-; CHECK-NEXT:  ret i1 %out
   %v1 = load i8, ptr %0, !range !1
   %v2 = load i8, ptr %1, !range !1
   %out = icmp eq i8 %v1, %v2
@@ -31,8 +22,6 @@ define i1 @cmp_different_range(ptr, ptr) {
 }
 
 define i1 @cmp_with_same_range(ptr, ptr) {
-; CHECK-LABEL: @cmp_with_same_range
-; CHECK: tail call i1 @cmp_with_range
   %v1 = load i8, ptr %0, !range !0
   %v2 = load i8, ptr %1, !range !0
   %out = icmp eq i8 %v1, %v2
@@ -42,3 +31,36 @@ define i1 @cmp_with_same_range(ptr, ptr) {
 ; The comparison must check every element of the range, not just the first pair.
 !0 = !{i8 0, i8 2, i8 21, i8 30}
 !1 = !{i8 0, i8 2, i8 21, i8 25}
+; CHECK-LABEL: define i1 @cmp_with_range(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1, !range [[RNG0:![0-9]+]]
+; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1, !range [[RNG0]]
+; CHECK-NEXT:    [[OUT:%.*]] = icmp eq i8 [[V1]], [[V2]]
+; CHECK-NEXT:    ret i1 [[OUT]]
+;
+;
+; CHECK-LABEL: define i1 @cmp_no_range(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1
+; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1
+; CHECK-NEXT:    [[OUT:%.*]] = icmp eq i8 [[V1]], [[V2]]
+; CHECK-NEXT:    ret i1 [[OUT]]
+;
+;
+; CHECK-LABEL: define i1 @cmp_different_range(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1, !range [[RNG1:![0-9]+]]
+; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1, !range [[RNG1]]
+; CHECK-NEXT:    [[OUT:%.*]] = icmp eq i8 [[V1]], [[V2]]
+; CHECK-NEXT:    ret i1 [[OUT]]
+;
+;
+; CHECK-LABEL: define i1 @cmp_with_same_range(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[TMP3:%.*]] = tail call i1 @cmp_with_range(ptr [[TMP0]], ptr [[TMP1]])
+; CHECK-NEXT:    ret i1 [[TMP3]]
+;
+;.
+; CHECK: [[RNG0]] = !{i8 0, i8 2, i8 21, i8 30}
+; CHECK: [[RNG1]] = !{i8 0, i8 2, i8 21, i8 25}
+;.
diff --git a/llvm/test/Transforms/MergeFunc/ranges.ll b/llvm/test/Transforms/MergeFunc/ranges.ll
index 94ed8eb7b0b07..e0e66506f2604 100644
--- a/llvm/test/Transforms/MergeFunc/ranges.ll
+++ b/llvm/test/Transforms/MergeFunc/ranges.ll
@@ -1,3 +1,4 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --include-generated-funcs --version 6
 ; RUN: opt -passes=mergefunc -S < %s | FileCheck %s
 define i1 @cmp_with_range(ptr, ptr) {
   %v1 = load i8, ptr %0, !range !0
@@ -7,11 +8,6 @@ define i1 @cmp_with_range(ptr, ptr) {
 }
 
 define i1 @cmp_no_range(ptr, ptr) {
-; CHECK-LABEL: @cmp_no_range
-; CHECK-NEXT: %v1 = load i8, ptr %0
-; CHECK-NEXT:  %v2 = load i8, ptr %1
-; CHECK-NEXT:  %out = icmp eq i8 %v1, %v2
-; CHECK-NEXT:  ret i1 %out
   %v1 = load i8, ptr %0
   %v2 = load i8, ptr %1
   %out = icmp eq i8 %v1, %v2
@@ -19,11 +15,6 @@ define i1 @cmp_no_range(ptr, ptr) {
 }
 
 define i1 @cmp_different_range(ptr, ptr) {
-; CHECK-LABEL: @cmp_different_range
-; CHECK-NEXT:  %v1 = load i8, ptr %0, align 1, !range !1
-; CHECK-NEXT:  %v2 = load i8, ptr %1, align 1, !range !1
-; CHECK-NEXT:  %out = icmp eq i8 %v1, %v2
-; CHECK-NEXT:  ret i1 %out
   %v1 = load i8, ptr %0, !range !1
   %v2 = load i8, ptr %1, !range !1
   %out = icmp eq i8 %v1, %v2
@@ -31,8 +22,6 @@ define i1 @cmp_different_range(ptr, ptr) {
 }
 
 define i1 @cmp_with_same_range(ptr, ptr) {
-; CHECK-LABEL: @cmp_with_same_range
-; CHECK: tail call i1 @cmp_with_range
   %v1 = load i8, ptr %0, !range !0
   %v2 = load i8, ptr %1, !range !0
   %out = icmp eq i8 %v1, %v2
@@ -41,3 +30,36 @@ define i1 @cmp_with_same_range(ptr, ptr) {
 
 !0 = !{i8 0, i8 2}
 !1 = !{i8 5, i8 7}
+; CHECK-LABEL: define i1 @cmp_with_range(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1, !range [[RNG0:![0-9]+]]
+; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1, !range [[RNG0]]
+; CHECK-NEXT:    [[OUT:%.*]] = icmp eq i8 [[V1]], [[V2]]
+; CHECK-NEXT:    ret i1 [[OUT]]
+;
+;
+; CHECK-LABEL: define i1 @cmp_no_range(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1
+; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1
+; CHECK-NEXT:    [[OUT:%.*]] = icmp eq i8 [[V1]], [[V2]]
+; CHECK-NEXT:    ret i1 [[OUT]]
+;
+;
+; CHECK-LABEL: define i1 @cmp_different_range(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1, !range [[RNG1:![0-9]+]]
+; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1, !range [[RNG1]]
+; CHECK-NEXT:    [[OUT:%.*]] = icmp eq i8 [[V1]], [[V2]]
+; CHECK-NEXT:    ret i1 [[OUT]]
+;
+;
+; CHECK-LABEL: define i1 @cmp_with_same_range(
+; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
+; CHECK-NEXT:    [[TMP3:%.*]] = tail call i1 @cmp_with_range(ptr [[TMP0]], ptr [[TMP1]])
+; CHECK-NEXT:    ret i1 [[TMP3]]
+;
+;.
+; CHECK: [[RNG0]] = !{i8 0, i8 2}
+; CHECK: [[RNG1]] = !{i8 5, i8 7}
+;.

>From 0d561ddf60cbc6e2f589271c27790bb13643a8b5 Mon Sep 17 00:00:00 2001
From: mmiftahx <mmiftah.duna at gmail.com>
Date: Wed, 23 Sep 2026 02:12:39 -0500
Subject: [PATCH 2/3] [MergeFunctions] Combine instruction metadata instead of
 comparing it

Instead of comparing instruction metadata in FunctionComparator, combine
it in MergeFunctions, as #220015 did for poison-generating flags.
FunctionComparator only compared metadata on loads and !range on calls,
and it treated nested nodes as equal. So the metadata of the kept
function, for example !tbaa on a store or !callees on a call, could be
wrong for the callers of the other function, while functions whose loads
differed only in !range or !nonnull were not merged.

The kept instruction also runs for the callers of the other function, so
use combineMetadataForCSE() with DoesKMove set. A few kinds need
different handling:
- !prof, !DIAssignID, !memprof, !callsite and !coro.outside.frame are
  left as they are on the kept instruction. Profile data is merged
  separately. !DIAssignID, !memprof and !callsite refer to their own
  function. Dropping !coro.outside.frame would let the alloca move into
  the coroutine frame.
- !tbaa.struct, !atomic.ignore.denormal.mode, !srcloc,
  !amdgpu.no.fine.grained.memory and !amdgpu.no.remote.memory are kept
  if both instructions have the same node. combineMetadataForCSE() drops
  them, which would make the merged function worse than either original.
  For example, some AMDGPU atomics would become CAS loops.
- !llvm.loop is kept if both loops have the same properties, ignoring
  debug locations. The loop IDs themselves are distinct nodes, so they
  almost never match.
- !llvm.mem.parallel_loop_access is dropped, because the same loop ID
  can belong to different loops in the two functions.
- !alias.scope and !noalias are dropped if the two functions declare
  different scopes with llvm.experimental.noalias.scope.decl, because
  the same attachment then means different things in each function.

Alias scopes and access groups are not mapped between the functions yet,
so they are usually dropped, even in functions that were already merged
before this change. For the same reason, loops with
llvm.loop.parallel_accesses lose their !llvm.loop. MergeFunctions runs
late in the pipeline, so this mostly affects codegen and full LTO.
---
 .../Transforms/Utils/FunctionComparator.h     |   6 +-
 llvm/lib/Transforms/IPO/MergeFunctions.cpp    |  93 ++++++++++++++
 .../Transforms/Utils/FunctionComparator.cpp   |  43 +------
 .../MergeFunc/call-and-invoke-with-ranges.ll  |  64 ++++------
 .../MergeFunc/instruction-metadata.ll         | 113 ++++++++----------
 .../MergeFunc/mergefunc-preserve-nonnull.ll   |  30 ++---
 .../Transforms/MergeFunc/ranges-multiple.ll   |  31 ++---
 llvm/test/Transforms/MergeFunc/ranges.ll      |  32 ++---
 8 files changed, 205 insertions(+), 207 deletions(-)

diff --git a/llvm/include/llvm/Transforms/Utils/FunctionComparator.h b/llvm/include/llvm/Transforms/Utils/FunctionComparator.h
index 8e1ca2dfeccb8..f07fda57aa435 100644
--- a/llvm/include/llvm/Transforms/Utils/FunctionComparator.h
+++ b/llvm/include/llvm/Transforms/Utils/FunctionComparator.h
@@ -90,7 +90,9 @@ class GlobalNumberState {
 /// FunctionComparator - Compares two functions to determine whether or not
 /// they will generate machine code with the same behaviour. DataLayout is
 /// used if available. The comparator always fails conservatively (erring on the
-/// side of claiming that two functions are different).
+/// side of claiming that two functions are different). Instruction flags, such
+/// as nuw or fast-math flags, and instruction metadata are not compared, so a
+/// user that merges two functions must combine them.
 class FunctionComparator {
 public:
   FunctionComparator(const Function *F1, const Function *F2,
@@ -266,7 +268,6 @@ class FunctionComparator {
   /// 6.2.Load: alignment (as integer numbers)
   /// 6.3.Load: ordering (as underlying enum class value)
   /// 6.4.Load: sync-scope (as integer numbers)
-  /// 6.5.Load: range metadata (as integer ranges)
   /// On this stage its better to see the code, since its not more than 10-15
   /// strings for particular instruction, and could change sometimes.
   ///
@@ -335,7 +336,6 @@ class FunctionComparator {
   int cmpAttrs(const AttributeList L, const AttributeList R) const;
   int cmpMDNode(const MDNode *L, const MDNode *R) const;
   int cmpMetadata(const Metadata *L, const Metadata *R) const;
-  int cmpInstMetadata(Instruction const *L, Instruction const *R) const;
   int cmpOperandBundlesSchema(const CallBase &LCS, const CallBase &RCS) const;
 
   /// Compare two GEPs for equivalent pointer arithmetic.
diff --git a/llvm/lib/Transforms/IPO/MergeFunctions.cpp b/llvm/lib/Transforms/IPO/MergeFunctions.cpp
index 7fc49a5fcc855..3bec6700ff5e1 100644
--- a/llvm/lib/Transforms/IPO/MergeFunctions.cpp
+++ b/llvm/lib/Transforms/IPO/MergeFunctions.cpp
@@ -130,6 +130,7 @@
 #include "llvm/Support/raw_ostream.h"
 #include "llvm/Transforms/IPO.h"
 #include "llvm/Transforms/Utils/FunctionComparator.h"
+#include "llvm/Transforms/Utils/Local.h"
 #include "llvm/Transforms/Utils/ModuleUtils.h"
 #include <algorithm>
 #include <cassert>
@@ -1083,6 +1084,68 @@ static void mergeValueProfileOnInstructions(Instruction *DstI,
                     VDs.size());
 }
 
+// Loop IDs are usually distinct nodes, so corresponding loops of the two
+// functions have different IDs even if they have the same properties. Compare
+// the property operands in order instead, ignoring the debug locations.
+static bool haveSameLoopProperties(const MDNode *A, const MDNode *B) {
+  // Loop::getLoopID() ignores a node whose first operand is not the node
+  // itself, so such a node gives the loop no properties.
+  auto IsLoopID = [](const MDNode *N) {
+    return N && N->getNumOperands() != 0 && N->getOperand(0) == N;
+  };
+  if (!IsLoopID(A) || !IsLoopID(B))
+    return A == B;
+  auto IsProperty = [](const MDOperand &Op) {
+    return !isa_and_present<DILocation>(Op.get());
+  };
+  return equal(make_filter_range(drop_begin(A->operands()), IsProperty),
+               make_filter_range(drop_begin(B->operands()), IsProperty));
+}
+
+// Combine the metadata of SrcI into DstI, so that it holds for the callers of
+// both functions. Metadata of the kinds in KeepIfSame is kept if it is the
+// same on both instructions.
+static void mergeMetadataOnInstructions(Instruction *DstI,
+                                        const Instruction *SrcI,
+                                        ArrayRef<unsigned> KeepIfSame) {
+  if (!DstI->hasMetadataOtherThanDebugLoc() &&
+      !SrcI->hasMetadataOtherThanDebugLoc())
+    return;
+
+  // Parallel loop accesses refer to loop IDs, and the same loop ID can belong
+  // to different loops in the two functions. Drop them, as loop IDs are not
+  // mapped between the functions.
+  DstI->setMetadata(LLVMContext::MD_mem_parallel_loop_access, nullptr);
+
+  // Keep the metadata of DstI that must not be combined: profile metadata is
+  // merged separately, assignment IDs and memprof call stacks refer to the
+  // function that contains them, and dropping !coro.outside.frame could move
+  // the alloca into the coroutine frame.
+  SmallVector<std::pair<unsigned, MDNode *>, 16> Kept;
+  for (unsigned Kind : {LLVMContext::MD_prof, LLVMContext::MD_DIAssignID,
+                        LLVMContext::MD_memprof, LLVMContext::MD_callsite,
+                        LLVMContext::MD_coro_outside_frame})
+    Kept.emplace_back(Kind, DstI->getMetadata(Kind));
+  for (unsigned Kind : KeepIfSame) {
+    MDNode *MD = DstI->getMetadata(Kind);
+    if (MD && MD == SrcI->getMetadata(Kind))
+      Kept.emplace_back(Kind, MD);
+  }
+  MDNode *DstLoop = DstI->getMetadata(LLVMContext::MD_loop);
+  if (haveSameLoopProperties(DstLoop, SrcI->getMetadata(LLVMContext::MD_loop)))
+    Kept.emplace_back(LLVMContext::MD_loop, DstLoop);
+
+  for (unsigned Kind : make_first_range(Kept))
+    DstI->setMetadata(Kind, nullptr);
+  // DstI also executes in place of SrcI, as if it had been moved there.
+  // TODO: Map the alias scopes and access groups of SrcI to those of DstI. They
+  // are usually different nodes, so they are dropped even if they correspond
+  // one to one.
+  combineMetadataForCSE(DstI, SrcI, /*DoesKMove=*/true);
+  for (auto [Kind, MD] : Kept)
+    DstI->setMetadata(Kind, MD);
+}
+
 void MergeFunctions::mergeInstrAnnotations(Function *Dst, Function *Src) {
   const BlockFrequencyInfo &DstBFI =
       FAM.getResult<BlockFrequencyAnalysis>(*Dst);
@@ -1094,11 +1157,41 @@ void MergeFunctions::mergeInstrAnnotations(Function *Dst, Function *Src) {
   // equivalent functions need not store their basic blocks in the same order.
   ReversePostOrderTraversal<Function *> DstRPOT(Dst);
   ReversePostOrderTraversal<Function *> SrcRPOT(Src);
+
+  // A noalias scope declaration sets how long its scope lasts, for example one
+  // loop iteration, and FunctionComparator does not compare the declared
+  // scopes. If they differ, the same !alias.scope or !noalias can mean
+  // different things in the two functions, so drop them.
+  bool SameScopeDecls = true;
+  for (auto [DstBB, SrcBB] : llvm::zip_equal(DstRPOT, SrcRPOT))
+    for (auto [DstI, SrcI] : llvm::zip_equal(*DstBB, *SrcBB))
+      if (auto *DstDecl = dyn_cast<NoAliasScopeDeclInst>(&DstI))
+        if (DstDecl->getScopeList() !=
+            cast<NoAliasScopeDeclInst>(SrcI).getScopeList())
+          SameScopeDecls = false;
+
+  // combineMetadataForCSE() drops these kinds, even if they are the same on
+  // both instructions. Losing them would make the merged function worse than
+  // both originals: memcpy would lose its TBAA, atomics would be expanded to
+  // compare-and-swap loops, and inline assembly diagnostics would lose their
+  // source location.
+  LLVMContext &Ctx = Dst->getContext();
+  const unsigned KeepIfSame[] = {
+      LLVMContext::MD_tbaa_struct, LLVMContext::MD_atomic_ignore_denormal_mode,
+      Ctx.getMDKindID("amdgpu.no.fine.grained.memory"),
+      Ctx.getMDKindID("amdgpu.no.remote.memory"), Ctx.getMDKindID("srcloc")};
+
   for (auto [DstBB, SrcBB] : llvm::zip_equal(DstRPOT, SrcRPOT)) {
     for (auto [DstI, SrcI] : llvm::zip_equal(*DstBB, *SrcBB)) {
       // Merge poison-generating flags.
       DstI.andIRFlags(&SrcI);
 
+      if (!SameScopeDecls) {
+        DstI.setMetadata(LLVMContext::MD_alias_scope, nullptr);
+        DstI.setMetadata(LLVMContext::MD_noalias, nullptr);
+      }
+      mergeMetadataOnInstructions(&DstI, &SrcI, KeepIfSame);
+
       MDNode *DstProf = DstI.getMetadata(LLVMContext::MD_prof);
       MDNode *SrcProf = SrcI.getMetadata(LLVMContext::MD_prof);
       if ((DstProf && isValueProfileMD(DstProf)) ||
diff --git a/llvm/lib/Transforms/Utils/FunctionComparator.cpp b/llvm/lib/Transforms/Utils/FunctionComparator.cpp
index 9770ae4e39f66..5e16dc3b4bdec 100644
--- a/llvm/lib/Transforms/Utils/FunctionComparator.cpp
+++ b/llvm/lib/Transforms/Utils/FunctionComparator.cpp
@@ -223,12 +223,6 @@ int FunctionComparator::cmpMDNode(const MDNode *L, const MDNode *R) const {
     return -1;
   if (!R)
     return 1;
-  // TODO: Note that as this is metadata, it is possible to drop and/or merge
-  // this data when considering functions to merge. Thus this comparison would
-  // return 0 (i.e. equivalent), but merging would become more complicated
-  // because the ranges would need to be unioned. It is not likely that
-  // functions differ ONLY in this metadata if they are actually the same
-  // function semantically.
   if (int Res = cmpNumbers(L->getNumOperands(), R->getNumOperands()))
     return Res;
   for (size_t I = 0; I < L->getNumOperands(); ++I)
@@ -237,29 +231,6 @@ int FunctionComparator::cmpMDNode(const MDNode *L, const MDNode *R) const {
   return 0;
 }
 
-int FunctionComparator::cmpInstMetadata(Instruction const *L,
-                                        Instruction const *R) const {
-  /// These metadata affects the other optimization passes by making assertions
-  /// or constraints.
-  /// Values that carry different expectations should be considered different.
-  SmallVector<std::pair<unsigned, MDNode *>> MDL, MDR;
-  L->getAllMetadataOtherThanDebugLoc(MDL);
-  R->getAllMetadataOtherThanDebugLoc(MDR);
-  if (MDL.size() > MDR.size())
-    return 1;
-  else if (MDL.size() < MDR.size())
-    return -1;
-  for (size_t I = 0, N = MDL.size(); I < N; ++I) {
-    auto const [KeyL, ML] = MDL[I];
-    auto const [KeyR, MR] = MDR[I];
-    if (int Res = cmpNumbers(KeyL, KeyR))
-      return Res;
-    if (int Res = cmpMDNode(ML, MR))
-      return Res;
-  }
-  return 0;
-}
-
 int FunctionComparator::cmpOperandBundlesSchema(const CallBase &LCS,
                                                 const CallBase &RCS) const {
   assert(LCS.getOpcode() == RCS.getOpcode() && "Can't compare otherwise!");
@@ -700,10 +671,8 @@ int FunctionComparator::cmpOperations(const Instruction *L,
     if (int Res =
             cmpOrderings(LI->getOrdering(), cast<LoadInst>(R)->getOrdering()))
       return Res;
-    if (int Res = cmpNumbers(LI->getSyncScopeID(),
-                             cast<LoadInst>(R)->getSyncScopeID()))
-      return Res;
-    return cmpInstMetadata(L, R);
+    return cmpNumbers(LI->getSyncScopeID(),
+                      cast<LoadInst>(R)->getSyncScopeID());
   }
   if (const StoreInst *SI = dyn_cast<StoreInst>(L)) {
     if (int Res =
@@ -731,11 +700,9 @@ int FunctionComparator::cmpOperations(const Instruction *L,
     if (int Res = cmpOperandBundlesSchema(*CBL, *CBR))
       return Res;
     if (const CallInst *CI = dyn_cast<CallInst>(L))
-      if (int Res = cmpNumbers(CI->getTailCallKind(),
-                               cast<CallInst>(R)->getTailCallKind()))
-        return Res;
-    return cmpMDNode(L->getMetadata(LLVMContext::MD_range),
-                     R->getMetadata(LLVMContext::MD_range));
+      return cmpNumbers(CI->getTailCallKind(),
+                        cast<CallInst>(R)->getTailCallKind());
+    return 0;
   }
   if (const SwitchInst *SI = dyn_cast<SwitchInst>(L)) {
     for (auto [LCase, RCase] : zip(SI->cases(), cast<SwitchInst>(R)->cases()))
diff --git a/llvm/test/Transforms/MergeFunc/call-and-invoke-with-ranges.ll b/llvm/test/Transforms/MergeFunc/call-and-invoke-with-ranges.ll
index 7ca189ceaa9a3..b32d26f5b153e 100644
--- a/llvm/test/Transforms/MergeFunc/call-and-invoke-with-ranges.ll
+++ b/llvm/test/Transforms/MergeFunc/call-and-invoke-with-ranges.ll
@@ -1,5 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --include-generated-funcs --version 6
-; RUN: opt -passes=mergefunc -S < %s | FileCheck %s
+; RUN: opt -passes=mergefunc -S < %s | FileCheck %s --implicit-check-not='!range'
+
+; Both groups merge, dropping ranges because one function in each has none.
 
 define i8 @call_with_range() {
   bitcast i8 0 to i8 ; dummy to make the function large enough
@@ -75,27 +77,15 @@ declare i32 @__gxx_personality_v0(...)
 
 !0 = !{i8 0, i8 2}
 !1 = !{i8 5, i8 7}
-; CHECK-LABEL: define i8 @call_with_range() {
-; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i8 0 to i8
-; CHECK-NEXT:    [[OUT:%.*]] = call i8 @dummy(), !range [[RNG0:![0-9]+]]
-; CHECK-NEXT:    ret i8 [[OUT]]
-;
-;
-; CHECK-LABEL: define i8 @call_no_range() {
-; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i8 0 to i8
-; CHECK-NEXT:    [[OUT:%.*]] = call i8 @dummy()
-; CHECK-NEXT:    ret i8 [[OUT]]
-;
-;
 ; CHECK-LABEL: define i8 @call_different_range() {
 ; CHECK-NEXT:    [[TMP1:%.*]] = bitcast i8 0 to i8
-; CHECK-NEXT:    [[OUT:%.*]] = call i8 @dummy(), !range [[RNG1:![0-9]+]]
+; CHECK-NEXT:    [[OUT:%.*]] = call i8 @dummy()
 ; CHECK-NEXT:    ret i8 [[OUT]]
 ;
 ;
-; CHECK-LABEL: define i8 @invoke_with_range() personality ptr undef {
+; CHECK-LABEL: define i8 @invoke_different_range() personality ptr undef {
 ; CHECK-NEXT:    [[OUT:%.*]] = invoke i8 @dummy()
-; CHECK-NEXT:            to label %[[NEXT:.*]] unwind label %[[LPAD:.*]], !range [[RNG0]]
+; CHECK-NEXT:            to label %[[NEXT:.*]] unwind label %[[LPAD:.*]]
 ; CHECK:       [[NEXT]]:
 ; CHECK-NEXT:    ret i8 [[OUT]]
 ; CHECK:       [[LPAD]]:
@@ -104,38 +94,32 @@ declare i32 @__gxx_personality_v0(...)
 ; CHECK-NEXT:    resume { ptr, i32 } zeroinitializer
 ;
 ;
-; CHECK-LABEL: define i8 @invoke_no_range() personality ptr undef {
-; CHECK-NEXT:    [[OUT:%.*]] = invoke i8 @dummy()
-; CHECK-NEXT:            to label %[[NEXT:.*]] unwind label %[[LPAD:.*]]
-; CHECK:       [[NEXT]]:
-; CHECK-NEXT:    ret i8 [[OUT]]
-; CHECK:       [[LPAD]]:
-; CHECK-NEXT:    [[PAD:%.*]] = landingpad { ptr, i32 }
-; CHECK-NEXT:            cleanup
-; CHECK-NEXT:    resume { ptr, i32 } zeroinitializer
+; CHECK-LABEL: define i8 @call_with_range() {
+; CHECK-NEXT:    [[TMP1:%.*]] = tail call i8 @call_different_range()
+; CHECK-NEXT:    ret i8 [[TMP1]]
 ;
 ;
-; CHECK-LABEL: define i8 @invoke_different_range() personality ptr undef {
-; CHECK-NEXT:    [[OUT:%.*]] = invoke i8 @dummy()
-; CHECK-NEXT:            to label %[[NEXT:.*]] unwind label %[[LPAD:.*]], !range [[RNG1]]
-; CHECK:       [[NEXT]]:
-; CHECK-NEXT:    ret i8 [[OUT]]
-; CHECK:       [[LPAD]]:
-; CHECK-NEXT:    [[PAD:%.*]] = landingpad { ptr, i32 }
-; CHECK-NEXT:            cleanup
-; CHECK-NEXT:    resume { ptr, i32 } zeroinitializer
+; CHECK-LABEL: define i8 @call_no_range() {
+; CHECK-NEXT:    [[TMP1:%.*]] = tail call i8 @call_different_range()
+; CHECK-NEXT:    ret i8 [[TMP1]]
 ;
 ;
 ; CHECK-LABEL: define i8 @call_with_same_range() {
-; CHECK-NEXT:    [[TMP1:%.*]] = tail call i8 @call_with_range()
+; CHECK-NEXT:    [[TMP1:%.*]] = tail call i8 @call_different_range()
+; CHECK-NEXT:    ret i8 [[TMP1]]
+;
+;
+; CHECK-LABEL: define i8 @invoke_with_range() personality ptr undef {
+; CHECK-NEXT:    [[TMP1:%.*]] = tail call i8 @invoke_different_range()
+; CHECK-NEXT:    ret i8 [[TMP1]]
+;
+;
+; CHECK-LABEL: define i8 @invoke_no_range() personality ptr undef {
+; CHECK-NEXT:    [[TMP1:%.*]] = tail call i8 @invoke_different_range()
 ; CHECK-NEXT:    ret i8 [[TMP1]]
 ;
 ;
 ; CHECK-LABEL: define i8 @invoke_with_same_range() personality ptr undef {
-; CHECK-NEXT:    [[TMP1:%.*]] = tail call i8 @invoke_with_range()
+; CHECK-NEXT:    [[TMP1:%.*]] = tail call i8 @invoke_different_range()
 ; CHECK-NEXT:    ret i8 [[TMP1]]
 ;
-;.
-; CHECK: [[RNG0]] = !{i8 0, i8 2}
-; CHECK: [[RNG1]] = !{i8 5, i8 7}
-;.
diff --git a/llvm/test/Transforms/MergeFunc/instruction-metadata.ll b/llvm/test/Transforms/MergeFunc/instruction-metadata.ll
index 327e8901f13f9..acfed631a8d37 100644
--- a/llvm/test/Transforms/MergeFunc/instruction-metadata.ll
+++ b/llvm/test/Transforms/MergeFunc/instruction-metadata.ll
@@ -1,5 +1,11 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --version 6
-; RUN: opt -S -passes=mergefunc < %s | FileCheck %s
+; RUN: opt -S -passes=mergefunc < %s | FileCheck %s \
+; RUN:   --implicit-check-not='!nonnull' --implicit-check-not='!callees' \
+; RUN:   --implicit-check-not='!amdgpu.no.fine.grained.memory' \
+; RUN:   --implicit-check-not='!alias.scope' --implicit-check-not='!noalias' \
+; RUN:   --implicit-check-not='!memprof' --implicit-check-not='!custom.kind' \
+; RUN:   --implicit-check-not='!llvm.loop' \
+; RUN:   --implicit-check-not='!llvm.mem.parallel_loop_access'
 
 declare i8 @f()
 declare void @callee()
@@ -23,13 +29,6 @@ define i8 @fn_range(ptr %p) {
 }
 
 define internal i8 @fn_range_b(ptr %p) {
-; CHECK-LABEL: define internal i8 @fn_range_b(
-; CHECK-SAME: ptr [[P:%.*]]) {
-; CHECK-NEXT:    [[V:%.*]] = load i8, ptr [[P]], align 1, !range [[RNG5:![0-9]+]], !noundef [[META4]]
-; CHECK-NEXT:    [[C:%.*]] = call i8 @f(), !range [[RNG5]]
-; CHECK-NEXT:    [[R:%.*]] = add i8 [[V]], [[C]]
-; CHECK-NEXT:    ret i8 [[R]]
-;
   %v = load i8, ptr %p, !range !1, !noundef !2
   %c = call i8 @f(), !range !1
   %r = add i8 %v, %c
@@ -39,7 +38,7 @@ define internal i8 @fn_range_b(ptr %p) {
 define ptr @fn_nonnull(ptr %p) {
 ; CHECK-LABEL: define ptr @fn_nonnull(
 ; CHECK-SAME: ptr [[P:%.*]]) {
-; CHECK-NEXT:    [[V:%.*]] = load ptr, ptr [[P]], align 8, !nonnull [[META4]]
+; CHECK-NEXT:    [[V:%.*]] = load ptr, ptr [[P]], align 8
 ; CHECK-NEXT:    ret ptr [[V]]
 ;
   %v = load ptr, ptr %p, !nonnull !2
@@ -47,11 +46,6 @@ define ptr @fn_nonnull(ptr %p) {
 }
 
 define internal ptr @fn_nonnull_b(ptr %p) {
-; CHECK-LABEL: define internal ptr @fn_nonnull_b(
-; CHECK-SAME: ptr [[P:%.*]]) {
-; CHECK-NEXT:    [[V:%.*]] = load ptr, ptr [[P]], align 8
-; CHECK-NEXT:    ret ptr [[V]]
-;
   %v = load ptr, ptr %p
   ret ptr %v
 }
@@ -59,7 +53,7 @@ define internal ptr @fn_nonnull_b(ptr %p) {
 define void @fn_tbaa(ptr %p) {
 ; CHECK-LABEL: define void @fn_tbaa(
 ; CHECK-SAME: ptr [[P:%.*]]) {
-; CHECK-NEXT:    store i32 0, ptr [[P]], align 4, !tbaa [[INT_TBAA6:![0-9]+]]
+; CHECK-NEXT:    store i32 0, ptr [[P]], align 4, !tbaa [[CHAR_TBAA5:![0-9]+]]
 ; CHECK-NEXT:    ret void
 ;
   store i32 0, ptr %p, !tbaa !3
@@ -74,7 +68,7 @@ define internal void @fn_tbaa_b(ptr %p) {
 define void @fn_callees(ptr %fp) {
 ; CHECK-LABEL: define void @fn_callees(
 ; CHECK-SAME: ptr [[FP:%.*]]) {
-; CHECK-NEXT:    call void [[FP]](), !callees [[META10:![0-9]+]]
+; CHECK-NEXT:    call void [[FP]]()
 ; CHECK-NEXT:    ret void
 ;
   call void %fp(), !callees !9
@@ -91,7 +85,7 @@ define internal void @fn_callees_b(ptr %fp) {
 define float @fn_atomic(ptr %p, float %v) {
 ; CHECK-LABEL: define float @fn_atomic(
 ; CHECK-SAME: ptr [[P:%.*]], float [[V:%.*]]) {
-; CHECK-NEXT:    [[R:%.*]] = atomicrmw fadd ptr [[P]], float [[V]] monotonic, align 4, !atomic.ignore.denormal.mode [[META4]], !amdgpu.no.fine.grained.memory [[META4]], !amdgpu.no.remote.memory [[META4]], !custom.kind [[META4]]
+; CHECK-NEXT:    [[R:%.*]] = atomicrmw fadd ptr [[P]], float [[V]] monotonic, align 4, !atomic.ignore.denormal.mode [[META4]], !amdgpu.no.remote.memory [[META4]]
 ; CHECK-NEXT:    [[S:%.*]] = atomicrmw fadd ptr [[P]], float [[R]] monotonic, align 4, !amdgpu.no.fine.grained.memory [[META4]]
 ; CHECK-NEXT:    ret float [[S]]
 ;
@@ -110,8 +104,8 @@ define internal float @fn_atomic_b(ptr %p, float %v) {
 define void @fn_same(ptr %p, ptr %q) {
 ; CHECK-LABEL: define void @fn_same(
 ; CHECK-SAME: ptr [[P:%.*]], ptr [[Q:%.*]]) {
-; CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i64(ptr [[P]], ptr [[Q]], i64 4, i1 false), !tbaa.struct [[TBAA_STRUCT11:![0-9]+]]
-; CHECK-NEXT:    call void asm sideeffect "", ""(), !srcloc [[META12:![0-9]+]]
+; CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i64(ptr [[P]], ptr [[Q]], i64 4, i1 false), !tbaa.struct [[TBAA_STRUCT8:![0-9]+]]
+; CHECK-NEXT:    call void asm sideeffect "", ""(), !srcloc [[META11:![0-9]+]]
 ; CHECK-NEXT:    ret void
 ;
   call void @llvm.memcpy.p0.p0.i64(ptr %p, ptr %q, i64 4, i1 false), !tbaa.struct !35
@@ -136,11 +130,11 @@ define void @fn_scope_decl(ptr %p, ptr %q, i64 %n) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    call void @llvm.experimental.noalias.scope.decl(metadata [[META13:![0-9]+]])
+; CHECK-NEXT:    call void @llvm.experimental.noalias.scope.decl(metadata [[META12:![0-9]+]])
 ; CHECK-NEXT:    [[PI:%.*]] = getelementptr i32, ptr [[P]], i64 [[I]]
-; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[PI]], align 4, !alias.scope [[META16:![0-9]+]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[PI]], align 4
 ; CHECK-NEXT:    [[QI:%.*]] = getelementptr i32, ptr [[Q]], i64 [[I]]
-; CHECK-NEXT:    store i32 [[V]], ptr [[QI]], align 4, !noalias [[META16]]
+; CHECK-NEXT:    store i32 [[V]], ptr [[QI]], align 4
 ; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 1
 ; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
 ; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
@@ -188,8 +182,8 @@ exit:
 define i32 @fn_scopes(ptr %a, ptr %b) {
 ; CHECK-LABEL: define i32 @fn_scopes(
 ; CHECK-SAME: ptr [[A:%.*]], ptr [[B:%.*]]) {
-; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[B]], align 4, !noalias [[META18:![0-9]+]]
-; CHECK-NEXT:    store i32 0, ptr [[A]], align 4, !alias.scope [[META18]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[B]], align 4, !noalias [[META4]]
+; CHECK-NEXT:    store i32 0, ptr [[A]], align 4
 ; CHECK-NEXT:    ret i32 [[V]]
 ;
   %v = load i32, ptr %b, !noalias !10
@@ -207,14 +201,14 @@ define internal i32 @fn_scopes_b(ptr %a, ptr %b) {
 ; debug locations of the loops may differ.
 define void @fn_loop(i64 %n) !dbg !29 {
 ; CHECK-LABEL: define void @fn_loop(
-; CHECK-SAME: i64 [[N:%.*]]) !dbg [[DBG21:![0-9]+]] {
+; CHECK-SAME: i64 [[N:%.*]]) !dbg [[DBG15:![0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 1
 ; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP23:![0-9]+]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP17:![0-9]+]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
@@ -254,7 +248,7 @@ define void @fn_loop_diff(i64 %n) {
 ; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 2
 ; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP27:![0-9]+]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
@@ -339,7 +333,7 @@ define void @fn_loop_invalid(i64 %n) {
 ; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 4
 ; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP28:![0-9]+]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
@@ -385,16 +379,16 @@ define void @fn_parallel(ptr %a, ptr %b) {
 ; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[OUTER]] ], [ [[I_NEXT:%.*]], %[[INNER]] ]
 ; CHECK-NEXT:    [[IDX:%.*]] = add i64 [[O]], [[I]]
 ; CHECK-NEXT:    [[PA:%.*]] = getelementptr i32, ptr [[A]], i64 [[IDX]]
-; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[PA]], align 4, !llvm.mem.parallel_loop_access [[META29:![0-9]+]]
+; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[PA]], align 4
 ; CHECK-NEXT:    [[PB:%.*]] = getelementptr i32, ptr [[B]], i64 [[IDX]]
-; CHECK-NEXT:    store i32 [[V]], ptr [[PB]], align 4, !llvm.mem.parallel_loop_access [[META29]]
+; CHECK-NEXT:    store i32 [[V]], ptr [[PB]], align 4
 ; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 1
 ; CHECK-NEXT:    [[I_DONE:%.*]] = icmp eq i64 [[I_NEXT]], 16
-; CHECK-NEXT:    br i1 [[I_DONE]], label %[[LATCH]], label %[[INNER]], !llvm.loop [[LOOP30:![0-9]+]]
+; CHECK-NEXT:    br i1 [[I_DONE]], label %[[LATCH]], label %[[INNER]], !llvm.loop [[LOOP21:![0-9]+]]
 ; CHECK:       [[LATCH]]:
 ; CHECK-NEXT:    [[O_NEXT]] = add i64 [[O]], 16
 ; CHECK-NEXT:    [[O_DONE:%.*]] = icmp eq i64 [[O_NEXT]], 64
-; CHECK-NEXT:    br i1 [[O_DONE]], label %[[EXIT:.*]], label %[[OUTER]], !llvm.loop [[LOOP31:![0-9]+]]
+; CHECK-NEXT:    br i1 [[O_DONE]], label %[[EXIT:.*]], label %[[OUTER]], !llvm.loop [[LOOP22:![0-9]+]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
@@ -476,7 +470,7 @@ define internal void @fn_coro_b() {
 define ptr @fn_memprof(i64 %n) {
 ; CHECK-LABEL: define ptr @fn_memprof(
 ; CHECK-SAME: i64 [[N:%.*]]) {
-; CHECK-NEXT:    [[P:%.*]] = call ptr @malloc(i64 [[N]]), !callsite [[META32:![0-9]+]]
+; CHECK-NEXT:    [[P:%.*]] = call ptr @malloc(i64 [[N]]), !callsite [[META23:![0-9]+]]
 ; CHECK-NEXT:    ret ptr [[P]]
 ;
   %p = call ptr @malloc(i64 %n), !callsite !20
@@ -492,9 +486,9 @@ define void @calls(ptr %p, i64 %n, float %v) {
 ; CHECK-LABEL: define void @calls(
 ; CHECK-SAME: ptr [[P:%.*]], i64 [[N:%.*]], float [[V:%.*]]) {
 ; CHECK-NEXT:    [[TMP1:%.*]] = call i8 @fn_range(ptr [[P]])
-; CHECK-NEXT:    [[TMP2:%.*]] = call i8 @fn_range_b(ptr [[P]])
+; CHECK-NEXT:    [[TMP2:%.*]] = call i8 @fn_range(ptr [[P]])
 ; CHECK-NEXT:    [[TMP3:%.*]] = call ptr @fn_nonnull(ptr [[P]])
-; CHECK-NEXT:    [[TMP4:%.*]] = call ptr @fn_nonnull_b(ptr [[P]])
+; CHECK-NEXT:    [[TMP4:%.*]] = call ptr @fn_nonnull(ptr [[P]])
 ; CHECK-NEXT:    call void @fn_tbaa(ptr [[P]])
 ; CHECK-NEXT:    call void @fn_tbaa(ptr [[P]])
 ; CHECK-NEXT:    call void @fn_callees(ptr [[P]])
@@ -614,34 +608,25 @@ define void @calls(ptr %p, i64 %n, float %v) {
 ; CHECK: [[META0:![0-9]+]] = distinct !DICompileUnit(language: DW_LANG_C11, file: [[META1:![0-9]+]], producer: "clang", isOptimized: true, runtimeVersion: 0, emissionKind: FullDebug)
 ; CHECK: [[META1]] = !DIFile(filename: "{{.*}}loop.c", directory: {{.*}})
 ; CHECK: [[META2:![0-9]+]] = !{i32 2, !"Debug Info Version", i32 3}
-; CHECK: [[RNG3]] = !{i8 0, i8 2}
+; CHECK: [[RNG3]] = !{i8 0, i8 2, i8 5, i8 7}
 ; CHECK: [[META4]] = !{}
-; CHECK: [[RNG5]] = !{i8 5, i8 7}
-; CHECK: [[INT_TBAA6]] = !{[[META7:![0-9]+]], [[META7]], i64 0}
-; CHECK: [[META7]] = !{!"int", [[META8:![0-9]+]], i64 0}
-; CHECK: [[META8]] = !{!"omnipotent char", [[META9:![0-9]+]], i64 0}
-; CHECK: [[META9]] = !{!"Simple C/C++ TBAA"}
-; CHECK: [[META10]] = !{ptr @callee}
-; CHECK: [[TBAA_STRUCT11]] = !{i64 0, i64 4, [[INT_TBAA6]]}
-; CHECK: [[META12]] = !{i64 42}
-; CHECK: [[META13]] = !{[[META14:![0-9]+]]}
-; CHECK: [[META14]] = distinct !{[[META14]], [[META15:![0-9]+]], !"other scope"}
-; CHECK: [[META15]] = distinct !{[[META15]], !"domain"}
-; CHECK: [[META16]] = !{[[META17:![0-9]+]]}
-; CHECK: [[META17]] = distinct !{[[META17]], [[META15]], !"scope"}
-; CHECK: [[META18]] = !{[[META19:![0-9]+]]}
-; CHECK: [[META19]] = distinct !{[[META19]], [[META20:![0-9]+]], !"scope"}
-; CHECK: [[META20]] = distinct !{[[META20]], !"domain"}
-; CHECK: [[DBG21]] = distinct !DISubprogram(name: "fn_loop", scope: [[META1]], file: [[META1]], line: 1, type: [[META22:![0-9]+]], scopeLine: 1, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META0]])
-; CHECK: [[META22]] = !DISubroutineType(types: [[META4]])
-; CHECK: [[LOOP23]] = distinct !{[[LOOP23]], [[META24:![0-9]+]], [[META25:![0-9]+]], [[META26:![0-9]+]]}
-; CHECK: [[META24]] = !DILocation(line: 2, column: 3, scope: [[DBG21]])
-; CHECK: [[META25]] = !DILocation(line: 3, column: 3, scope: [[DBG21]])
-; CHECK: [[META26]] = !{!"llvm.loop.mustprogress"}
-; CHECK: [[LOOP27]] = distinct !{[[LOOP27]], [[META26]]}
-; CHECK: [[LOOP28]] = distinct !{[[LOOP28]], [[META26]]}
-; CHECK: [[META29]] = !{[[LOOP30]]}
-; CHECK: [[LOOP30]] = distinct !{[[LOOP30]]}
-; CHECK: [[LOOP31]] = distinct !{[[LOOP31]]}
-; CHECK: [[META32]] = !{i64 111}
+; CHECK: [[CHAR_TBAA5]] = !{[[META6:![0-9]+]], [[META6]], i64 0}
+; CHECK: [[META6]] = !{!"omnipotent char", [[META7:![0-9]+]], i64 0}
+; CHECK: [[META7]] = !{!"Simple C/C++ TBAA"}
+; CHECK: [[TBAA_STRUCT8]] = !{i64 0, i64 4, [[META9:![0-9]+]]}
+; CHECK: [[META9]] = !{[[META10:![0-9]+]], [[META10]], i64 0}
+; CHECK: [[META10]] = !{!"int", [[META6]], i64 0}
+; CHECK: [[META11]] = !{i64 42}
+; CHECK: [[META12]] = !{[[META13:![0-9]+]]}
+; CHECK: [[META13]] = distinct !{[[META13]], [[META14:![0-9]+]], !"other scope"}
+; CHECK: [[META14]] = distinct !{[[META14]], !"domain"}
+; CHECK: [[DBG15]] = distinct !DISubprogram(name: "fn_loop", scope: [[META1]], file: [[META1]], line: 1, type: [[META16:![0-9]+]], scopeLine: 1, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META0]])
+; CHECK: [[META16]] = !DISubroutineType(types: [[META4]])
+; CHECK: [[LOOP17]] = distinct !{[[LOOP17]], [[META18:![0-9]+]], [[META19:![0-9]+]], [[META20:![0-9]+]]}
+; CHECK: [[META18]] = !DILocation(line: 2, column: 3, scope: [[DBG15]])
+; CHECK: [[META19]] = !DILocation(line: 3, column: 3, scope: [[DBG15]])
+; CHECK: [[META20]] = !{!"llvm.loop.mustprogress"}
+; CHECK: [[LOOP21]] = distinct !{[[LOOP21]]}
+; CHECK: [[LOOP22]] = distinct !{[[LOOP22]]}
+; CHECK: [[META23]] = !{i64 111}
 ;.
diff --git a/llvm/test/Transforms/MergeFunc/mergefunc-preserve-nonnull.ll b/llvm/test/Transforms/MergeFunc/mergefunc-preserve-nonnull.ll
index 329378dc705c0..d05a3240819d4 100644
--- a/llvm/test/Transforms/MergeFunc/mergefunc-preserve-nonnull.ll
+++ b/llvm/test/Transforms/MergeFunc/mergefunc-preserve-nonnull.ll
@@ -1,8 +1,10 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --include-generated-funcs --version 6
-; RUN: opt -passes=mergefunc -S < %s | FileCheck %s
+; RUN: opt -passes=mergefunc -S < %s | FileCheck %s --implicit-check-not='!nonnull' \
+; RUN:   --implicit-check-not='!noundef' --implicit-check-not='!noalias' \
+; RUN:   --implicit-check-not='!alias.scope'
 
-; This test makes sure that the mergefunc pass does not merge functions
-; that have different nonnull assertions.
+; Merge functions with different load and store metadata, dropping assertions
+; that do not hold for every body.
 
 %1 = type ptr
 
@@ -36,7 +38,6 @@ define void @noundef_dbg(ptr %0, ptr %1) {
   ret void
 }
 
-; FIXME: This is merged despite different noalias metadata.
 define void @noalias_2(ptr %0, ptr %1) {
   %3 = load ptr, ptr %1, align 8, !noalias !7
   store ptr %3, ptr %0, align 8, !alias.scope !7
@@ -54,46 +55,37 @@ define void @noalias_2(ptr %0, ptr %1) {
 !8 = !DILocation(line: 1, column: 1, scope: !{})
 ; CHECK-LABEL: define void @f1(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1]], align 8, !nonnull [[META0:![0-9]+]]
+; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1]], align 8
 ; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0]], align 8
 ; CHECK-NEXT:    ret void
 ;
 ;
 ; CHECK-LABEL: define void @f2(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1]], align 8
-; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0]], align 8
+; CHECK-NEXT:    tail call void @f1(ptr [[TMP0]], ptr [[TMP1]])
 ; CHECK-NEXT:    ret void
 ;
 ;
 ; CHECK-LABEL: define void @noundef(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1]], align 8, !noundef [[META0]]
-; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0]], align 8
+; CHECK-NEXT:    tail call void @f1(ptr [[TMP0]], ptr [[TMP1]])
 ; CHECK-NEXT:    ret void
 ;
 ;
 ; CHECK-LABEL: define void @noalias_1(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    [[TMP3:%.*]] = load ptr, ptr [[TMP1]], align 8, !noalias [[META1:![0-9]+]]
-; CHECK-NEXT:    store ptr [[TMP3]], ptr [[TMP0]], align 8, !alias.scope [[META1]]
+; CHECK-NEXT:    tail call void @f1(ptr [[TMP0]], ptr [[TMP1]])
 ; CHECK-NEXT:    ret void
 ;
 ;
 ; CHECK-LABEL: define void @noundef_dbg(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    tail call void @noundef(ptr [[TMP0]], ptr [[TMP1]])
+; CHECK-NEXT:    tail call void @f1(ptr [[TMP0]], ptr [[TMP1]])
 ; CHECK-NEXT:    ret void
 ;
 ;
 ; CHECK-LABEL: define void @noalias_2(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    tail call void @noalias_1(ptr [[TMP0]], ptr [[TMP1]])
+; CHECK-NEXT:    tail call void @f1(ptr [[TMP0]], ptr [[TMP1]])
 ; CHECK-NEXT:    ret void
 ;
-;.
-; CHECK: [[META0]] = !{}
-; CHECK: [[META1]] = !{[[META2:![0-9]+]]}
-; CHECK: [[META2]] = distinct !{[[META2]], [[META3:![0-9]+]]}
-; CHECK: [[META3]] = distinct !{[[META3]]}
-;.
diff --git a/llvm/test/Transforms/MergeFunc/ranges-multiple.ll b/llvm/test/Transforms/MergeFunc/ranges-multiple.ll
index 198a457361208..1275f1e92b23c 100644
--- a/llvm/test/Transforms/MergeFunc/ranges-multiple.ll
+++ b/llvm/test/Transforms/MergeFunc/ranges-multiple.ll
@@ -1,5 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --include-generated-funcs --version 6
 ; RUN: opt -passes=mergefunc -S < %s | FileCheck %s
+
+; The shared body gets the union of the ranges, with every interval.
 define i1 @cmp_with_range(ptr, ptr) {
   %v1 = load i8, ptr %0, !range !0
   %v2 = load i8, ptr %1, !range !0
@@ -7,13 +9,6 @@ define i1 @cmp_with_range(ptr, ptr) {
   ret i1 %out
 }
 
-define i1 @cmp_no_range(ptr, ptr) {
-  %v1 = load i8, ptr %0
-  %v2 = load i8, ptr %1
-  %out = icmp eq i8 %v1, %v2
-  ret i1 %out
-}
-
 define i1 @cmp_different_range(ptr, ptr) {
   %v1 = load i8, ptr %0, !range !1
   %v2 = load i8, ptr %1, !range !1
@@ -28,10 +23,9 @@ define i1 @cmp_with_same_range(ptr, ptr) {
   ret i1 %out
 }
 
-; The comparison must check every element of the range, not just the first pair.
 !0 = !{i8 0, i8 2, i8 21, i8 30}
 !1 = !{i8 0, i8 2, i8 21, i8 25}
-; CHECK-LABEL: define i1 @cmp_with_range(
+; CHECK-LABEL: define i1 @cmp_different_range(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
 ; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1, !range [[RNG0:![0-9]+]]
 ; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1, !range [[RNG0]]
@@ -39,28 +33,17 @@ define i1 @cmp_with_same_range(ptr, ptr) {
 ; CHECK-NEXT:    ret i1 [[OUT]]
 ;
 ;
-; CHECK-LABEL: define i1 @cmp_no_range(
-; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1
-; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1
-; CHECK-NEXT:    [[OUT:%.*]] = icmp eq i8 [[V1]], [[V2]]
-; CHECK-NEXT:    ret i1 [[OUT]]
-;
-;
-; CHECK-LABEL: define i1 @cmp_different_range(
+; CHECK-LABEL: define i1 @cmp_with_range(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1, !range [[RNG1:![0-9]+]]
-; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1, !range [[RNG1]]
-; CHECK-NEXT:    [[OUT:%.*]] = icmp eq i8 [[V1]], [[V2]]
-; CHECK-NEXT:    ret i1 [[OUT]]
+; CHECK-NEXT:    [[TMP3:%.*]] = tail call i1 @cmp_different_range(ptr [[TMP0]], ptr [[TMP1]])
+; CHECK-NEXT:    ret i1 [[TMP3]]
 ;
 ;
 ; CHECK-LABEL: define i1 @cmp_with_same_range(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    [[TMP3:%.*]] = tail call i1 @cmp_with_range(ptr [[TMP0]], ptr [[TMP1]])
+; CHECK-NEXT:    [[TMP3:%.*]] = tail call i1 @cmp_different_range(ptr [[TMP0]], ptr [[TMP1]])
 ; CHECK-NEXT:    ret i1 [[TMP3]]
 ;
 ;.
 ; CHECK: [[RNG0]] = !{i8 0, i8 2, i8 21, i8 30}
-; CHECK: [[RNG1]] = !{i8 0, i8 2, i8 21, i8 25}
 ;.
diff --git a/llvm/test/Transforms/MergeFunc/ranges.ll b/llvm/test/Transforms/MergeFunc/ranges.ll
index e0e66506f2604..159e57f1899f9 100644
--- a/llvm/test/Transforms/MergeFunc/ranges.ll
+++ b/llvm/test/Transforms/MergeFunc/ranges.ll
@@ -1,5 +1,7 @@
 ; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --check-globals all --include-generated-funcs --version 6
-; RUN: opt -passes=mergefunc -S < %s | FileCheck %s
+; RUN: opt -passes=mergefunc -S < %s | FileCheck %s --implicit-check-not='!range'
+
+; The shared body loses its range because one function has no restriction.
 define i1 @cmp_with_range(ptr, ptr) {
   %v1 = load i8, ptr %0, !range !0
   %v2 = load i8, ptr %1, !range !0
@@ -30,36 +32,28 @@ define i1 @cmp_with_same_range(ptr, ptr) {
 
 !0 = !{i8 0, i8 2}
 !1 = !{i8 5, i8 7}
-; CHECK-LABEL: define i1 @cmp_with_range(
+; CHECK-LABEL: define i1 @cmp_different_range(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1, !range [[RNG0:![0-9]+]]
-; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1, !range [[RNG0]]
+; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1
+; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1
 ; CHECK-NEXT:    [[OUT:%.*]] = icmp eq i8 [[V1]], [[V2]]
 ; CHECK-NEXT:    ret i1 [[OUT]]
 ;
 ;
-; CHECK-LABEL: define i1 @cmp_no_range(
+; CHECK-LABEL: define i1 @cmp_with_range(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1
-; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1
-; CHECK-NEXT:    [[OUT:%.*]] = icmp eq i8 [[V1]], [[V2]]
-; CHECK-NEXT:    ret i1 [[OUT]]
+; CHECK-NEXT:    [[TMP3:%.*]] = tail call i1 @cmp_different_range(ptr [[TMP0]], ptr [[TMP1]])
+; CHECK-NEXT:    ret i1 [[TMP3]]
 ;
 ;
-; CHECK-LABEL: define i1 @cmp_different_range(
+; CHECK-LABEL: define i1 @cmp_no_range(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    [[V1:%.*]] = load i8, ptr [[TMP0]], align 1, !range [[RNG1:![0-9]+]]
-; CHECK-NEXT:    [[V2:%.*]] = load i8, ptr [[TMP1]], align 1, !range [[RNG1]]
-; CHECK-NEXT:    [[OUT:%.*]] = icmp eq i8 [[V1]], [[V2]]
-; CHECK-NEXT:    ret i1 [[OUT]]
+; CHECK-NEXT:    [[TMP3:%.*]] = tail call i1 @cmp_different_range(ptr [[TMP0]], ptr [[TMP1]])
+; CHECK-NEXT:    ret i1 [[TMP3]]
 ;
 ;
 ; CHECK-LABEL: define i1 @cmp_with_same_range(
 ; CHECK-SAME: ptr [[TMP0:%.*]], ptr [[TMP1:%.*]]) {
-; CHECK-NEXT:    [[TMP3:%.*]] = tail call i1 @cmp_with_range(ptr [[TMP0]], ptr [[TMP1]])
+; CHECK-NEXT:    [[TMP3:%.*]] = tail call i1 @cmp_different_range(ptr [[TMP0]], ptr [[TMP1]])
 ; CHECK-NEXT:    ret i1 [[TMP3]]
 ;
-;.
-; CHECK: [[RNG0]] = !{i8 0, i8 2}
-; CHECK: [[RNG1]] = !{i8 5, i8 7}
-;.

>From baafae6aed7fdd0c4dbe8ab386bad06a8574ec6c Mon Sep 17 00:00:00 2001
From: mmiftahx <mmiftah.duna at gmail.com>
Date: Tue, 29 Sep 2026 22:51:45 -0500
Subject: [PATCH 3/3] Remove KeepIfSame

---
 llvm/lib/Transforms/IPO/MergeFunctions.cpp    | 24 +-------
 .../MergeFunc/instruction-metadata.ll         | 57 +++++++++----------
 2 files changed, 31 insertions(+), 50 deletions(-)

diff --git a/llvm/lib/Transforms/IPO/MergeFunctions.cpp b/llvm/lib/Transforms/IPO/MergeFunctions.cpp
index 3bec6700ff5e1..a3b97654c83b2 100644
--- a/llvm/lib/Transforms/IPO/MergeFunctions.cpp
+++ b/llvm/lib/Transforms/IPO/MergeFunctions.cpp
@@ -1103,11 +1103,9 @@ static bool haveSameLoopProperties(const MDNode *A, const MDNode *B) {
 }
 
 // Combine the metadata of SrcI into DstI, so that it holds for the callers of
-// both functions. Metadata of the kinds in KeepIfSame is kept if it is the
-// same on both instructions.
+// both functions.
 static void mergeMetadataOnInstructions(Instruction *DstI,
-                                        const Instruction *SrcI,
-                                        ArrayRef<unsigned> KeepIfSame) {
+                                        const Instruction *SrcI) {
   if (!DstI->hasMetadataOtherThanDebugLoc() &&
       !SrcI->hasMetadataOtherThanDebugLoc())
     return;
@@ -1126,11 +1124,6 @@ static void mergeMetadataOnInstructions(Instruction *DstI,
                         LLVMContext::MD_memprof, LLVMContext::MD_callsite,
                         LLVMContext::MD_coro_outside_frame})
     Kept.emplace_back(Kind, DstI->getMetadata(Kind));
-  for (unsigned Kind : KeepIfSame) {
-    MDNode *MD = DstI->getMetadata(Kind);
-    if (MD && MD == SrcI->getMetadata(Kind))
-      Kept.emplace_back(Kind, MD);
-  }
   MDNode *DstLoop = DstI->getMetadata(LLVMContext::MD_loop);
   if (haveSameLoopProperties(DstLoop, SrcI->getMetadata(LLVMContext::MD_loop)))
     Kept.emplace_back(LLVMContext::MD_loop, DstLoop);
@@ -1170,17 +1163,6 @@ void MergeFunctions::mergeInstrAnnotations(Function *Dst, Function *Src) {
             cast<NoAliasScopeDeclInst>(SrcI).getScopeList())
           SameScopeDecls = false;
 
-  // combineMetadataForCSE() drops these kinds, even if they are the same on
-  // both instructions. Losing them would make the merged function worse than
-  // both originals: memcpy would lose its TBAA, atomics would be expanded to
-  // compare-and-swap loops, and inline assembly diagnostics would lose their
-  // source location.
-  LLVMContext &Ctx = Dst->getContext();
-  const unsigned KeepIfSame[] = {
-      LLVMContext::MD_tbaa_struct, LLVMContext::MD_atomic_ignore_denormal_mode,
-      Ctx.getMDKindID("amdgpu.no.fine.grained.memory"),
-      Ctx.getMDKindID("amdgpu.no.remote.memory"), Ctx.getMDKindID("srcloc")};
-
   for (auto [DstBB, SrcBB] : llvm::zip_equal(DstRPOT, SrcRPOT)) {
     for (auto [DstI, SrcI] : llvm::zip_equal(*DstBB, *SrcBB)) {
       // Merge poison-generating flags.
@@ -1190,7 +1172,7 @@ void MergeFunctions::mergeInstrAnnotations(Function *Dst, Function *Src) {
         DstI.setMetadata(LLVMContext::MD_alias_scope, nullptr);
         DstI.setMetadata(LLVMContext::MD_noalias, nullptr);
       }
-      mergeMetadataOnInstructions(&DstI, &SrcI, KeepIfSame);
+      mergeMetadataOnInstructions(&DstI, &SrcI);
 
       MDNode *DstProf = DstI.getMetadata(LLVMContext::MD_prof);
       MDNode *SrcProf = SrcI.getMetadata(LLVMContext::MD_prof);
diff --git a/llvm/test/Transforms/MergeFunc/instruction-metadata.ll b/llvm/test/Transforms/MergeFunc/instruction-metadata.ll
index acfed631a8d37..b2cd48b8a98d4 100644
--- a/llvm/test/Transforms/MergeFunc/instruction-metadata.ll
+++ b/llvm/test/Transforms/MergeFunc/instruction-metadata.ll
@@ -2,6 +2,9 @@
 ; RUN: opt -S -passes=mergefunc < %s | FileCheck %s \
 ; RUN:   --implicit-check-not='!nonnull' --implicit-check-not='!callees' \
 ; RUN:   --implicit-check-not='!amdgpu.no.fine.grained.memory' \
+; RUN:   --implicit-check-not='!amdgpu.no.remote.memory' \
+; RUN:   --implicit-check-not='!atomic.ignore.denormal.mode' \
+; RUN:   --implicit-check-not='!tbaa.struct' --implicit-check-not='!srcloc' \
 ; RUN:   --implicit-check-not='!alias.scope' --implicit-check-not='!noalias' \
 ; RUN:   --implicit-check-not='!memprof' --implicit-check-not='!custom.kind' \
 ; RUN:   --implicit-check-not='!llvm.loop' \
@@ -80,13 +83,13 @@ define internal void @fn_callees_b(ptr %fp) {
   ret void
 }
 
-; AMDGPU atomic metadata and !atomic.ignore.denormal.mode are kept if they are
-; the same in both functions. Other unknown metadata is dropped.
+; These kinds are dropped even if they are the same in both functions, because
+; combineMetadataForCSE() doesn't handle them.
 define float @fn_atomic(ptr %p, float %v) {
 ; CHECK-LABEL: define float @fn_atomic(
 ; CHECK-SAME: ptr [[P:%.*]], float [[V:%.*]]) {
-; CHECK-NEXT:    [[R:%.*]] = atomicrmw fadd ptr [[P]], float [[V]] monotonic, align 4, !atomic.ignore.denormal.mode [[META4]], !amdgpu.no.remote.memory [[META4]]
-; CHECK-NEXT:    [[S:%.*]] = atomicrmw fadd ptr [[P]], float [[R]] monotonic, align 4, !amdgpu.no.fine.grained.memory [[META4]]
+; CHECK-NEXT:    [[R:%.*]] = atomicrmw fadd ptr [[P]], float [[V]] monotonic, align 4
+; CHECK-NEXT:    [[S:%.*]] = atomicrmw fadd ptr [[P]], float [[R]] monotonic, align 4
 ; CHECK-NEXT:    ret float [[S]]
 ;
   %r = atomicrmw fadd ptr %p, float %v monotonic, !amdgpu.no.fine.grained.memory !2, !amdgpu.no.remote.memory !2, !atomic.ignore.denormal.mode !2, !custom.kind !2
@@ -100,12 +103,12 @@ define internal float @fn_atomic_b(ptr %p, float %v) {
   ret float %s
 }
 
-; !tbaa.struct and !srcloc are kept if they are the same in both functions.
+; The same goes for !tbaa.struct and !srcloc on calls.
 define void @fn_same(ptr %p, ptr %q) {
 ; CHECK-LABEL: define void @fn_same(
 ; CHECK-SAME: ptr [[P:%.*]], ptr [[Q:%.*]]) {
-; CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i64(ptr [[P]], ptr [[Q]], i64 4, i1 false), !tbaa.struct [[TBAA_STRUCT8:![0-9]+]]
-; CHECK-NEXT:    call void asm sideeffect "", ""(), !srcloc [[META11:![0-9]+]]
+; CHECK-NEXT:    call void @llvm.memcpy.p0.p0.i64(ptr [[P]], ptr [[Q]], i64 4, i1 false)
+; CHECK-NEXT:    call void asm sideeffect "", ""()
 ; CHECK-NEXT:    ret void
 ;
   call void @llvm.memcpy.p0.p0.i64(ptr %p, ptr %q, i64 4, i1 false), !tbaa.struct !35
@@ -130,7 +133,7 @@ define void @fn_scope_decl(ptr %p, ptr %q, i64 %n) {
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
-; CHECK-NEXT:    call void @llvm.experimental.noalias.scope.decl(metadata [[META12:![0-9]+]])
+; CHECK-NEXT:    call void @llvm.experimental.noalias.scope.decl(metadata [[META8:![0-9]+]])
 ; CHECK-NEXT:    [[PI:%.*]] = getelementptr i32, ptr [[P]], i64 [[I]]
 ; CHECK-NEXT:    [[V:%.*]] = load i32, ptr [[PI]], align 4
 ; CHECK-NEXT:    [[QI:%.*]] = getelementptr i32, ptr [[Q]], i64 [[I]]
@@ -201,14 +204,14 @@ define internal i32 @fn_scopes_b(ptr %a, ptr %b) {
 ; debug locations of the loops may differ.
 define void @fn_loop(i64 %n) !dbg !29 {
 ; CHECK-LABEL: define void @fn_loop(
-; CHECK-SAME: i64 [[N:%.*]]) !dbg [[DBG15:![0-9]+]] {
+; CHECK-SAME: i64 [[N:%.*]]) !dbg [[DBG11:![0-9]+]] {
 ; CHECK-NEXT:  [[ENTRY:.*]]:
 ; CHECK-NEXT:    br label %[[LOOP:.*]]
 ; CHECK:       [[LOOP]]:
 ; CHECK-NEXT:    [[I:%.*]] = phi i64 [ 0, %[[ENTRY]] ], [ [[I_NEXT:%.*]], %[[LOOP]] ]
 ; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 1
 ; CHECK-NEXT:    [[DONE:%.*]] = icmp eq i64 [[I_NEXT]], [[N]]
-; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP17:![0-9]+]]
+; CHECK-NEXT:    br i1 [[DONE]], label %[[EXIT:.*]], label %[[LOOP]], !llvm.loop [[LOOP13:![0-9]+]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
@@ -384,11 +387,11 @@ define void @fn_parallel(ptr %a, ptr %b) {
 ; CHECK-NEXT:    store i32 [[V]], ptr [[PB]], align 4
 ; CHECK-NEXT:    [[I_NEXT]] = add i64 [[I]], 1
 ; CHECK-NEXT:    [[I_DONE:%.*]] = icmp eq i64 [[I_NEXT]], 16
-; CHECK-NEXT:    br i1 [[I_DONE]], label %[[LATCH]], label %[[INNER]], !llvm.loop [[LOOP21:![0-9]+]]
+; CHECK-NEXT:    br i1 [[I_DONE]], label %[[LATCH]], label %[[INNER]], !llvm.loop [[LOOP17:![0-9]+]]
 ; CHECK:       [[LATCH]]:
 ; CHECK-NEXT:    [[O_NEXT]] = add i64 [[O]], 16
 ; CHECK-NEXT:    [[O_DONE:%.*]] = icmp eq i64 [[O_NEXT]], 64
-; CHECK-NEXT:    br i1 [[O_DONE]], label %[[EXIT:.*]], label %[[OUTER]], !llvm.loop [[LOOP22:![0-9]+]]
+; CHECK-NEXT:    br i1 [[O_DONE]], label %[[EXIT:.*]], label %[[OUTER]], !llvm.loop [[LOOP18:![0-9]+]]
 ; CHECK:       [[EXIT]]:
 ; CHECK-NEXT:    ret void
 ;
@@ -470,7 +473,7 @@ define internal void @fn_coro_b() {
 define ptr @fn_memprof(i64 %n) {
 ; CHECK-LABEL: define ptr @fn_memprof(
 ; CHECK-SAME: i64 [[N:%.*]]) {
-; CHECK-NEXT:    [[P:%.*]] = call ptr @malloc(i64 [[N]]), !callsite [[META23:![0-9]+]]
+; CHECK-NEXT:    [[P:%.*]] = call ptr @malloc(i64 [[N]]), !callsite [[META19:![0-9]+]]
 ; CHECK-NEXT:    ret ptr [[P]]
 ;
   %p = call ptr @malloc(i64 %n), !callsite !20
@@ -613,20 +616,16 @@ define void @calls(ptr %p, i64 %n, float %v) {
 ; CHECK: [[CHAR_TBAA5]] = !{[[META6:![0-9]+]], [[META6]], i64 0}
 ; CHECK: [[META6]] = !{!"omnipotent char", [[META7:![0-9]+]], i64 0}
 ; CHECK: [[META7]] = !{!"Simple C/C++ TBAA"}
-; CHECK: [[TBAA_STRUCT8]] = !{i64 0, i64 4, [[META9:![0-9]+]]}
-; CHECK: [[META9]] = !{[[META10:![0-9]+]], [[META10]], i64 0}
-; CHECK: [[META10]] = !{!"int", [[META6]], i64 0}
-; CHECK: [[META11]] = !{i64 42}
-; CHECK: [[META12]] = !{[[META13:![0-9]+]]}
-; CHECK: [[META13]] = distinct !{[[META13]], [[META14:![0-9]+]], !"other scope"}
-; CHECK: [[META14]] = distinct !{[[META14]], !"domain"}
-; CHECK: [[DBG15]] = distinct !DISubprogram(name: "fn_loop", scope: [[META1]], file: [[META1]], line: 1, type: [[META16:![0-9]+]], scopeLine: 1, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META0]])
-; CHECK: [[META16]] = !DISubroutineType(types: [[META4]])
-; CHECK: [[LOOP17]] = distinct !{[[LOOP17]], [[META18:![0-9]+]], [[META19:![0-9]+]], [[META20:![0-9]+]]}
-; CHECK: [[META18]] = !DILocation(line: 2, column: 3, scope: [[DBG15]])
-; CHECK: [[META19]] = !DILocation(line: 3, column: 3, scope: [[DBG15]])
-; CHECK: [[META20]] = !{!"llvm.loop.mustprogress"}
-; CHECK: [[LOOP21]] = distinct !{[[LOOP21]]}
-; CHECK: [[LOOP22]] = distinct !{[[LOOP22]]}
-; CHECK: [[META23]] = !{i64 111}
+; CHECK: [[META8]] = !{[[META9:![0-9]+]]}
+; CHECK: [[META9]] = distinct !{[[META9]], [[META10:![0-9]+]], !"other scope"}
+; CHECK: [[META10]] = distinct !{[[META10]], !"domain"}
+; CHECK: [[DBG11]] = distinct !DISubprogram(name: "fn_loop", scope: [[META1]], file: [[META1]], line: 1, type: [[META12:![0-9]+]], scopeLine: 1, spFlags: DISPFlagDefinition | DISPFlagOptimized, unit: [[META0]])
+; CHECK: [[META12]] = !DISubroutineType(types: [[META4]])
+; CHECK: [[LOOP13]] = distinct !{[[LOOP13]], [[META14:![0-9]+]], [[META15:![0-9]+]], [[META16:![0-9]+]]}
+; CHECK: [[META14]] = !DILocation(line: 2, column: 3, scope: [[DBG11]])
+; CHECK: [[META15]] = !DILocation(line: 3, column: 3, scope: [[DBG11]])
+; CHECK: [[META16]] = !{!"llvm.loop.mustprogress"}
+; CHECK: [[LOOP17]] = distinct !{[[LOOP17]]}
+; CHECK: [[LOOP18]] = distinct !{[[LOOP18]]}
+; CHECK: [[META19]] = !{i64 111}
 ;.



More information about the llvm-commits mailing list