[llvm] [BOLT] Make .debug_names emission reproducible (PR #225218)

Rafael Auler via llvm-commits llvm-commits at lists.llvm.org
Tue Sep 22 17:53:09 PDT 2026


https://github.com/rafaelauler updated https://github.com/llvm/llvm-project/pull/225218

>From d12baec56a90148db5f7538cdda9d2bbe6439c57 Mon Sep 17 00:00:00 2001
From: Rafael Auler <rafaelauler at fb.com>
Date: Fri, 18 Sep 2026 16:27:49 -0700
Subject: [PATCH 1/2] [BOLT] Make .debug_names emission reproducible

Summary:
--update-debug-sections does not produce the same binary twice when the
input has a .debug_names section, or is given one with
--create-debug-names-section. Three sections vary from run to run:
.debug_str, .debug_names and .debug_str_offsets. On a two-CU
split-DWARF test binary, ten runs of the same llvm-bolt produced three
different outputs.

PR #197859 made the per-bucket work merge in partition order precisely so
the output would be reproducible, but missed this section. Two issues:

 - .debug_str offsets were handed out from the workers, so main.dwo.dwo
   added by the merge step lands in a different place in each
   run. The string offsets stored in .debug_names follow from it, and
   so does .debug_str_offsets;
 - hash colliding names were ordered by insertion (worker) order, also
   non-deterministic.

We now determine string positions in the non-concurrent finalize step(),
and order collisions.
---
 bolt/include/bolt/Core/DebugNames.h |  6 ++++++
 bolt/lib/Core/DebugNames.cpp        | 25 +++++++++++++++++++++----
 2 files changed, 27 insertions(+), 4 deletions(-)

diff --git a/bolt/include/bolt/Core/DebugNames.h b/bolt/include/bolt/Core/DebugNames.h
index 4213cfc7283b60..a87d4c3c67b2c6 100644
--- a/bolt/include/bolt/Core/DebugNames.h
+++ b/bolt/include/bolt/Core/DebugNames.h
@@ -109,6 +109,10 @@ class DWARF5AcceleratorTable {
     uint32_t HashValue;
     uint64_t EntryOffset;
     std::vector<BOLTDWARF5AccelTableData *> Values;
+    /// Points at the name this entry was keyed by. Invalid before finalize()
+    const std::string *Name = nullptr;
+    /// New name (not in .debug_str). StrOffset is invalid before finalize()
+    bool NeedsStrOffset = false;
   };
   using HashList = std::vector<HashData *>;
   using BucketList = std::vector<HashList>;
@@ -170,6 +174,8 @@ class DWARF5AcceleratorTable {
   getSecondIndexForEntry(const BOLTDWARF5AccelTableData &Value) const;
   /// Uniquify Entries.
   void finalize();
+  /// Append the new names that are not already in .debug_str
+  void assignStrOffsets();
   /// Computes bucket count.
   void computeBucketCount();
   /// Populate Abbreviations Map.
diff --git a/bolt/lib/Core/DebugNames.cpp b/bolt/lib/Core/DebugNames.cpp
index dc2494beca1530..3b06f90390a349 100644
--- a/bolt/lib/Core/DebugNames.cpp
+++ b/bolt/lib/Core/DebugNames.cpp
@@ -334,13 +334,15 @@ std::optional<std::string> DWARF5AcceleratorTable::getName(
       // Need to find offset for the name in the .debug_str section.
       llvm::hash_code Hash = llvm::hash_value(llvm::StringRef(Name));
       auto ItCache = StrCacheToOffsetMap.find(Hash);
+      // New string not in the input, we'll assign an offset later.
       if (ItCache == StrCacheToOffsetMap.end())
-        NameIndexOffset = MainBinaryStrWriter.addString(Name);
+        It.NeedsStrOffset = true;
       else
         NameIndexOffset = ItCache->second;
     }
+    // New string not in the input, we'll assign an offset later.
     if (!NameToUse.empty())
-      NameIndexOffset = MainBinaryStrWriter.addString(Name);
+      It.NeedsStrOffset = true;
     It.StrOffset = NameIndexOffset;
     // This is the same hash function used in DWARF5AccelTableData.
     It.HashValue = caseFoldingDjbHash(Name);
@@ -546,9 +548,11 @@ void DWARF5AcceleratorTable::finalize() {
   // data structures.
   computeBucketCount();
 
-  // Compute bucket contents and final ordering.
+  // Compute bucket contents and final ordering. Entries is done growing, so
+  // its keys have settled and an entry can point back at its own name.
   Buckets.resize(BucketCount);
   for (auto &E : Entries) {
+    E.second.Name = &E.first;
     uint32_t Bucket = E.second.HashValue % BucketCount;
     Buckets[Bucket].push_back(&E.second);
   }
@@ -557,7 +561,8 @@ void DWARF5AcceleratorTable::finalize() {
   // up together. Stable sort makes testing easier and doesn't cost much more.
   for (HashList &Bucket : Buckets) {
     llvm::stable_sort(Bucket, [](const HashData *LHS, const HashData *RHS) {
-      return LHS->HashValue < RHS->HashValue;
+      return std::tie(LHS->HashValue, *LHS->Name) <
+             std::tie(RHS->HashValue, *RHS->Name);
     });
     for (HashData *H : Bucket)
       llvm::stable_sort(H->Values, [](const BOLTDWARF5AccelTableData *LHS,
@@ -571,6 +576,8 @@ void DWARF5AcceleratorTable::finalize() {
       });
   }
 
+  assignStrOffsets();
+
   CUIndexForm = DIEInteger::BestForm(/*IsSigned*/ false, CUList.size() - 1);
   TUIndexForm = DIEInteger::BestForm(
       /*IsSigned*/ false, LocalTUList.size() + ForeignTUList.size() - 1);
@@ -579,6 +586,16 @@ void DWARF5AcceleratorTable::finalize() {
   TUIndexEncodingSize = *dwarf::getFixedFormByteSize(TUIndexForm, FormParams);
 }
 
+void DWARF5AcceleratorTable::assignStrOffsets() {
+  // Walk the buckets, which finalize() has just put in a total order.
+  for (HashList &Bucket : Buckets)
+    for (HashData *Hash : Bucket)
+      if (Hash->NeedsStrOffset) {
+        assert(Hash->Name && "Name back-pointer was not set");
+        Hash->StrOffset = MainBinaryStrWriter.addString(*Hash->Name);
+      }
+}
+
 std::optional<DWARF5AccelTable::UnitIndexAndEncoding>
 DWARF5AcceleratorTable::getIndexForEntry(
     const BOLTDWARF5AccelTableData &Value) const {

>From a3cba3c620f0089111715f83de6d732a0de3734a Mon Sep 17 00:00:00 2001
From: Rafael Auler <rafaelauler at fb.com>
Date: Tue, 22 Sep 2026 17:51:15 -0700
Subject: [PATCH 2/2] Close a loose end for synthesized debug (anonymous
 namespace) strings that happen to be already in the main CU debug_str
 section.

---
 bolt/lib/Core/DebugNames.cpp | 11 ++++-------
 1 file changed, 4 insertions(+), 7 deletions(-)

diff --git a/bolt/lib/Core/DebugNames.cpp b/bolt/lib/Core/DebugNames.cpp
index 3b06f90390a349..829200e9d182ce 100644
--- a/bolt/lib/Core/DebugNames.cpp
+++ b/bolt/lib/Core/DebugNames.cpp
@@ -329,20 +329,17 @@ std::optional<std::string> DWARF5AcceleratorTable::getName(
   }
   auto &It = Entries[Name];
   if (It.Values.empty()) {
-    if (DWOID && NameToUse.empty()) {
-      // For DWO Unit the offset is in the .debug_str.dwo section.
-      // Need to find offset for the name in the .debug_str section.
+    if (DWOID || !NameToUse.empty()) {
+      // The offset in hand is into .debug_str.dwo, or absent for a synthesized
+      // name. Look for its offset in the main .debug_str.
       llvm::hash_code Hash = llvm::hash_value(llvm::StringRef(Name));
       auto ItCache = StrCacheToOffsetMap.find(Hash);
-      // New string not in the input, we'll assign an offset later.
+      // New string not in the main .debug_str, we'll assign an offset later.
       if (ItCache == StrCacheToOffsetMap.end())
         It.NeedsStrOffset = true;
       else
         NameIndexOffset = ItCache->second;
     }
-    // New string not in the input, we'll assign an offset later.
-    if (!NameToUse.empty())
-      It.NeedsStrOffset = true;
     It.StrOffset = NameIndexOffset;
     // This is the same hash function used in DWARF5AccelTableData.
     It.HashValue = caseFoldingDjbHash(Name);



More information about the llvm-commits mailing list