[llvm] [IR] Preserve TBAA metadata when merging equivalent memory accesses (PR #208499)
via llvm-commits
llvm-commits at lists.llvm.org
Thu Jul 9 09:32:47 PDT 2026
https://github.com/222rohan created https://github.com/llvm/llvm-project/pull/208499
### Description
When CSE/GVN merge two equivalent memory accesses (e.g. two loads of the same
address) and only one carries `!tbaa`, whether the survivor keeps the tag
depends on source order: `combineMetadata` only looks at the survivor's
metadata, so if the survivor is untyped the removed access's tag is dropped.
This handles TBAA after the metadata loop (like `prof`/`mmra`/`memprof`/
`callsite`): keep the existing merge when the survivor has a tag, and copy the
removed access's tag when it is the only one, so the result is order-independent.
### Example
```llvm
; opt -passes=early-cse -S
define i32 @untyped_first(ptr %p) {
%b = load i32, ptr %p, align 4 ; untyped, becomes survivor
%a = load i32, ptr %p, align 4, !tbaa !0 ; typed, removed
%s = add i32 %a, %b
ret i32 %s
}
```
Before: survivor loses `!tbaa`. After: survivor keeps `!tbaa !0` (same as the
typed-first ordering).
### Microbenchmark
The dropped tag prevents TBAA from proving no-alias, which blocks LICM. In this
kernel a loop-invariant division chain is hoisted only when the tag survives
(`clang -O1`):
```c
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <time.h>
__attribute__((always_inline))
static inline float div_chain(float x, float d) {
x = x / d;
x = x / d;
x = x / d;
x = x / d;
return x;
}
// Typed *p read first -> survives CSE -> load keeps !tbaa "int".
__attribute__((noinline))
float order_tbaa_first(float *arr, int *p, int n) {
for (int i = 0; i < n; i++) {
int a = *p; // typed load: load i32, !tbaa "int"
int b; memcpy(&b, p, sizeof(int)); // memcpy load: CSE'd away (second)
float d = (float)(a + b);
float r = div_chain(1.0e30f, d); // function of *p only
arr[i] = r;
}
return arr[n - 1];
}
// memcpy read first -> survives CSE -> load has no !tbaa.
__attribute__((noinline))
float order_tbaa_second(float *arr, int *p, int n) {
for (int i = 0; i < n; i++) {
int b; memcpy(&b, p, sizeof(int)); // memcpy load: load i32 (no tbaa, first)
int a = *p; // typed load: CSE'd away (second)
float d = (float)(a + b);
float r = div_chain(1.0e30f, d);
arr[i] = r;
}
return arr[n - 1];
}
static double now_sec(void) {
struct timespec ts;
clock_gettime(CLOCK_MONOTONIC, &ts);
return (double)ts.tv_sec + (double)ts.tv_nsec * 1e-9;
}
int main(int argc, char **argv) {
int n = argc > 1 ? atoi(argv[1]) : 200 * 1000 * 1000;
int reps = argc > 2 ? atoi(argv[2]) : 7;
float *arr = malloc(sizeof(float) * (size_t)n);
if (!arr) { perror("malloc"); return 1; }
int val = 3;
double bt = 1e18, bm = 1e18;
float r1 = 0, r2 = 0;
for (int r = 0; r < reps; r++) {
double t0 = now_sec();
r1 = order_tbaa_first(arr, &val, n);
double dt = now_sec() - t0;
if (dt < bt) bt = dt;
}
for (int r = 0; r < reps; r++) {
double t0 = now_sec();
r2 = order_tbaa_second(arr, &val, n);
double dt = now_sec() - t0;
if (dt < bm) bm = dt;
}
printf("n = %d, reps = %d \n", n, reps);
printf("order_tbaa_first (tbaa kept -> chain hoisted by LICM) result=%.3e best=%.4f s (%.2f ns/iter)\n",
r1, bt, bt * 1e9 / n);
printf("order_tbaa_second (tbaa dropped -> chain per iteration) result=%.3e best=%.4f s (%.2f ns/iter)\n",
r2, bm, bm * 1e9 / n);
printf("speedup (second / first) = %.2fx\n", bm / bt);
free(arr);
return 0;
}
```
(Generated with the help of Cursor)
| | survivor tag | div chain | time |
|---|---|---|---|
| tag kept (fixed) | `!tbaa` | hoisted out of loop | 0.22 ns/iter |
| tag dropped (before) | none | recomputed every iteration | 2.11 ns/iter |
**9.6x** on this kernel (0.22 vs 2.11 ns/iter, `n=2e8`).
Setup: AMD Ryzen 9 7950X, 32GB RAM, Ubuntu 22.04.5 (kernel 6.8.0), clang built
from this branch, `clang -O1` (no `-ffast-math`). The gap is specific to `-O1`.
>From bd04d6adc6c1b762a7ed82f0cf45172a1ff1b802 Mon Sep 17 00:00:00 2001
From: Rohan Shenoy <rshenoy at amd.com>
Date: Thu, 9 Jul 2026 17:38:31 +0530
Subject: [PATCH] [IR] Preserve TBAA metadata when merging equivalent memory
accesses
When CSE/GVN merge two equivalent memory accesses (e.g. two loads of the same
address) into one, the surviving instruction should keep the !tbaa tag if
either access had one. Currently combineMetadata only keeps TBAA when the
surviving instruction already had a tag, so if the survivor is untyped the tag
from the removed access is dropped. This makes the result depend on source
order (which access becomes the survivor).
---
llvm/lib/Transforms/Utils/Local.cpp | 27 +++++++++----
.../Transforms/EarlyCSE/tbaa-merge-order.ll | 39 +++++++++++++++++++
2 files changed, 59 insertions(+), 7 deletions(-)
create mode 100644 llvm/test/Transforms/EarlyCSE/tbaa-merge-order.ll
diff --git a/llvm/lib/Transforms/Utils/Local.cpp b/llvm/lib/Transforms/Utils/Local.cpp
index b17740c0bc192..b067dbfb5ac22 100644
--- a/llvm/lib/Transforms/Utils/Local.cpp
+++ b/llvm/lib/Transforms/Utils/Local.cpp
@@ -2964,10 +2964,6 @@ static void combineMetadata(Instruction *K, const Instruction *J,
if (!AAOnly)
K->mergeDIAssignID(J);
break;
- case LLVMContext::MD_tbaa:
- if (DoesKMove)
- K->setMetadata(Kind, MDNode::getMostGenericTBAA(JMD, KMD));
- break;
case LLVMContext::MD_alias_scope:
if (DoesKMove)
K->setMetadata(Kind, MDNode::getMostGenericAliasScope(JMD, KMD));
@@ -3007,9 +3003,10 @@ static void combineMetadata(Instruction *K, const Instruction *J,
if (!AAOnly && (DoesKMove || !K->hasMetadata(LLVMContext::MD_noundef)))
K->setMetadata(Kind, JMD);
break;
- // Keep empty cases for prof, mmra, memprof, and callsite to prevent them
- // from being removed as unknown metadata. The actual merging is handled
- // separately below.
+ // Keep empty cases for tbaa, prof, mmra, memprof, and callsite to prevent
+ // them from being removed as unknown metadata. The actual merging is
+ // handled separately below.
+ case LLVMContext::MD_tbaa:
case LLVMContext::MD_prof:
case LLVMContext::MD_mmra:
case LLVMContext::MD_memprof:
@@ -3113,6 +3110,22 @@ static void combineMetadata(Instruction *K, const Instruction *J,
K->setMetadata(LLVMContext::MD_prof,
MDNode::getMergedProfMetadata(KProf, JProf, K, J));
}
+
+ // Merge TBAA metadata.
+ // Handle separately to support cases where only one instruction has the
+ // metadata.
+ MDNode *JTBAA = J->getMetadata(LLVMContext::MD_tbaa);
+ MDNode *KTBAA = K->getMetadata(LLVMContext::MD_tbaa);
+ if (KTBAA) {
+ if (DoesKMove)
+ K->setMetadata(LLVMContext::MD_tbaa,
+ MDNode::getMostGenericTBAA(JTBAA, KTBAA));
+ } else if (!AAOnly && JTBAA &&
+ isa<LoadInst, StoreInst, CallInst, VAArgInst, AtomicRMWInst,
+ AtomicCmpXchgInst>(K)) {
+ // Only J has a tag: copy it so the result is independent of CSE order.
+ K->setMetadata(LLVMContext::MD_tbaa, JTBAA);
+ }
}
void llvm::combineMetadataForCSE(Instruction *K, const Instruction *J,
diff --git a/llvm/test/Transforms/EarlyCSE/tbaa-merge-order.ll b/llvm/test/Transforms/EarlyCSE/tbaa-merge-order.ll
new file mode 100644
index 0000000000000..a3f742a490eec
--- /dev/null
+++ b/llvm/test/Transforms/EarlyCSE/tbaa-merge-order.ll
@@ -0,0 +1,39 @@
+; NOTE: Assertions have been autogenerated by utils/update_test_checks.py UTC_ARGS: --version 5
+; RUN: opt -passes=early-cse -S < %s | FileCheck %s
+
+; Two loads of the same address are CSE'd into one. Only one of them carries
+; !tbaa. The surviving load's metadata must not depend on which load was
+; written first in the source (i.e. which becomes the CSE survivor).
+
+; typed (tbaa) load first, untyped load second.
+define i32 @typed_first(ptr %p) {
+; CHECK-LABEL: define i32 @typed_first(
+; CHECK-SAME: ptr [[P:%.*]]) {
+; CHECK-NEXT: [[A:%.*]] = load i32, ptr [[P]], align 4, !tbaa [[INT_TBAA0:![0-9]+]]
+; CHECK-NEXT: [[S:%.*]] = add i32 [[A]], [[A]]
+; CHECK-NEXT: ret i32 [[S]]
+;
+ %a = load i32, ptr %p, align 4, !tbaa !0
+ %b = load i32, ptr %p, align 4
+ %s = add i32 %a, %b
+ ret i32 %s
+}
+
+; untyped load first, typed (tbaa) load second. Result must match typed_first.
+define i32 @untyped_first(ptr %p) {
+; CHECK-LABEL: define i32 @untyped_first(
+; CHECK-SAME: ptr [[P:%.*]]) {
+; CHECK-NEXT: [[B:%.*]] = load i32, ptr [[P]], align 4, !tbaa [[INT_TBAA0]]
+; CHECK-NEXT: [[S:%.*]] = add i32 [[B]], [[B]]
+; CHECK-NEXT: ret i32 [[S]]
+;
+ %b = load i32, ptr %p, align 4
+ %a = load i32, ptr %p, align 4, !tbaa !0
+ %s = add i32 %a, %b
+ ret i32 %s
+}
+
+!0 = !{!1, !1, i64 0}
+!1 = !{!"int", !2, i64 0}
+!2 = !{!"omnipotent char", !3, i64 0}
+!3 = !{!"Simple C/C++ TBAA"}
More information about the llvm-commits
mailing list