[llvm] [BOLT] Make .debug_names emission reproducible (PR #225218)
Rafael Auler via llvm-commits
llvm-commits at lists.llvm.org
Tue Sep 22 17:53:09 PDT 2026
https://github.com/rafaelauler updated https://github.com/llvm/llvm-project/pull/225218
>From d12baec56a90148db5f7538cdda9d2bbe6439c57 Mon Sep 17 00:00:00 2001
From: Rafael Auler <rafaelauler at fb.com>
Date: Fri, 18 Sep 2026 16:27:49 -0700
Subject: [PATCH 1/2] [BOLT] Make .debug_names emission reproducible
Summary:
--update-debug-sections does not produce the same binary twice when the
input has a .debug_names section, or is given one with
--create-debug-names-section. Three sections vary from run to run:
.debug_str, .debug_names and .debug_str_offsets. On a two-CU
split-DWARF test binary, ten runs of the same llvm-bolt produced three
different outputs.
PR #197859 made the per-bucket work merge in partition order precisely so
the output would be reproducible, but missed this section. Two issues:
- .debug_str offsets were handed out from the workers, so main.dwo.dwo
added by the merge step lands in a different place in each
run. The string offsets stored in .debug_names follow from it, and
so does .debug_str_offsets;
- hash colliding names were ordered by insertion (worker) order, also
non-deterministic.
We now determine string positions in the non-concurrent finalize step(),
and order collisions.
---
bolt/include/bolt/Core/DebugNames.h | 6 ++++++
bolt/lib/Core/DebugNames.cpp | 25 +++++++++++++++++++++----
2 files changed, 27 insertions(+), 4 deletions(-)
diff --git a/bolt/include/bolt/Core/DebugNames.h b/bolt/include/bolt/Core/DebugNames.h
index 4213cfc7283b60..a87d4c3c67b2c6 100644
--- a/bolt/include/bolt/Core/DebugNames.h
+++ b/bolt/include/bolt/Core/DebugNames.h
@@ -109,6 +109,10 @@ class DWARF5AcceleratorTable {
uint32_t HashValue;
uint64_t EntryOffset;
std::vector<BOLTDWARF5AccelTableData *> Values;
+ /// Points at the name this entry was keyed by. Invalid before finalize()
+ const std::string *Name = nullptr;
+ /// New name (not in .debug_str). StrOffset is invalid before finalize()
+ bool NeedsStrOffset = false;
};
using HashList = std::vector<HashData *>;
using BucketList = std::vector<HashList>;
@@ -170,6 +174,8 @@ class DWARF5AcceleratorTable {
getSecondIndexForEntry(const BOLTDWARF5AccelTableData &Value) const;
/// Uniquify Entries.
void finalize();
+ /// Append the new names that are not already in .debug_str
+ void assignStrOffsets();
/// Computes bucket count.
void computeBucketCount();
/// Populate Abbreviations Map.
diff --git a/bolt/lib/Core/DebugNames.cpp b/bolt/lib/Core/DebugNames.cpp
index dc2494beca1530..3b06f90390a349 100644
--- a/bolt/lib/Core/DebugNames.cpp
+++ b/bolt/lib/Core/DebugNames.cpp
@@ -334,13 +334,15 @@ std::optional<std::string> DWARF5AcceleratorTable::getName(
// Need to find offset for the name in the .debug_str section.
llvm::hash_code Hash = llvm::hash_value(llvm::StringRef(Name));
auto ItCache = StrCacheToOffsetMap.find(Hash);
+ // New string not in the input, we'll assign an offset later.
if (ItCache == StrCacheToOffsetMap.end())
- NameIndexOffset = MainBinaryStrWriter.addString(Name);
+ It.NeedsStrOffset = true;
else
NameIndexOffset = ItCache->second;
}
+ // New string not in the input, we'll assign an offset later.
if (!NameToUse.empty())
- NameIndexOffset = MainBinaryStrWriter.addString(Name);
+ It.NeedsStrOffset = true;
It.StrOffset = NameIndexOffset;
// This is the same hash function used in DWARF5AccelTableData.
It.HashValue = caseFoldingDjbHash(Name);
@@ -546,9 +548,11 @@ void DWARF5AcceleratorTable::finalize() {
// data structures.
computeBucketCount();
- // Compute bucket contents and final ordering.
+ // Compute bucket contents and final ordering. Entries is done growing, so
+ // its keys have settled and an entry can point back at its own name.
Buckets.resize(BucketCount);
for (auto &E : Entries) {
+ E.second.Name = &E.first;
uint32_t Bucket = E.second.HashValue % BucketCount;
Buckets[Bucket].push_back(&E.second);
}
@@ -557,7 +561,8 @@ void DWARF5AcceleratorTable::finalize() {
// up together. Stable sort makes testing easier and doesn't cost much more.
for (HashList &Bucket : Buckets) {
llvm::stable_sort(Bucket, [](const HashData *LHS, const HashData *RHS) {
- return LHS->HashValue < RHS->HashValue;
+ return std::tie(LHS->HashValue, *LHS->Name) <
+ std::tie(RHS->HashValue, *RHS->Name);
});
for (HashData *H : Bucket)
llvm::stable_sort(H->Values, [](const BOLTDWARF5AccelTableData *LHS,
@@ -571,6 +576,8 @@ void DWARF5AcceleratorTable::finalize() {
});
}
+ assignStrOffsets();
+
CUIndexForm = DIEInteger::BestForm(/*IsSigned*/ false, CUList.size() - 1);
TUIndexForm = DIEInteger::BestForm(
/*IsSigned*/ false, LocalTUList.size() + ForeignTUList.size() - 1);
@@ -579,6 +586,16 @@ void DWARF5AcceleratorTable::finalize() {
TUIndexEncodingSize = *dwarf::getFixedFormByteSize(TUIndexForm, FormParams);
}
+void DWARF5AcceleratorTable::assignStrOffsets() {
+ // Walk the buckets, which finalize() has just put in a total order.
+ for (HashList &Bucket : Buckets)
+ for (HashData *Hash : Bucket)
+ if (Hash->NeedsStrOffset) {
+ assert(Hash->Name && "Name back-pointer was not set");
+ Hash->StrOffset = MainBinaryStrWriter.addString(*Hash->Name);
+ }
+}
+
std::optional<DWARF5AccelTable::UnitIndexAndEncoding>
DWARF5AcceleratorTable::getIndexForEntry(
const BOLTDWARF5AccelTableData &Value) const {
>From a3cba3c620f0089111715f83de6d732a0de3734a Mon Sep 17 00:00:00 2001
From: Rafael Auler <rafaelauler at fb.com>
Date: Tue, 22 Sep 2026 17:51:15 -0700
Subject: [PATCH 2/2] Close a loose end for synthesized debug (anonymous
namespace) strings that happen to be already in the main CU debug_str
section.
---
bolt/lib/Core/DebugNames.cpp | 11 ++++-------
1 file changed, 4 insertions(+), 7 deletions(-)
diff --git a/bolt/lib/Core/DebugNames.cpp b/bolt/lib/Core/DebugNames.cpp
index 3b06f90390a349..829200e9d182ce 100644
--- a/bolt/lib/Core/DebugNames.cpp
+++ b/bolt/lib/Core/DebugNames.cpp
@@ -329,20 +329,17 @@ std::optional<std::string> DWARF5AcceleratorTable::getName(
}
auto &It = Entries[Name];
if (It.Values.empty()) {
- if (DWOID && NameToUse.empty()) {
- // For DWO Unit the offset is in the .debug_str.dwo section.
- // Need to find offset for the name in the .debug_str section.
+ if (DWOID || !NameToUse.empty()) {
+ // The offset in hand is into .debug_str.dwo, or absent for a synthesized
+ // name. Look for its offset in the main .debug_str.
llvm::hash_code Hash = llvm::hash_value(llvm::StringRef(Name));
auto ItCache = StrCacheToOffsetMap.find(Hash);
- // New string not in the input, we'll assign an offset later.
+ // New string not in the main .debug_str, we'll assign an offset later.
if (ItCache == StrCacheToOffsetMap.end())
It.NeedsStrOffset = true;
else
NameIndexOffset = ItCache->second;
}
- // New string not in the input, we'll assign an offset later.
- if (!NameToUse.empty())
- It.NeedsStrOffset = true;
It.StrOffset = NameIndexOffset;
// This is the same hash function used in DWARF5AccelTableData.
It.HashValue = caseFoldingDjbHash(Name);
More information about the llvm-commits
mailing list