[llvm] [utils] Fix DenseMap debugger printers for the packed used-bit array (PR #201755)

Fangrui Song via llvm-commits llvm-commits at lists.llvm.org
Sun Jun 7 19:13:48 PDT 2026


https://github.com/MaskRay updated https://github.com/llvm/llvm-project/pull/201755

>From dd9a8e81b1a8a9772109225d104f56b91aa01fbd Mon Sep 17 00:00:00 2001
From: Fangrui Song <i at maskray.me>
Date: Thu, 4 Jun 2026 22:41:08 -0700
Subject: [PATCH 1/3] [utils] Fix DenseMap debugger printers for the packed
 used-bit array

DenseMap no longer use in-band sentinel keys. (#200595 and #201281).
Update the GDB pretty printer and LLDB data formatters to test the used
bit rather than comparing keys.

GDB: advancePastEmptyBuckets relied on DenseMapInfo::getEmptyKey(), which
could not be evaluated in GDB and so was disabled, leaving the printer to
emit empty and erased buckets. It now walks bucket indices and skips any
whose used bit is clear.

LLDB: DenseMapSynthetic used a key-uniqueness heuristic to guess which
buckets were live, which mishandled a lone erased bucket (hence the
former tombstones=1 summary note). It now reads the used array directly,
so erased entries are skipped exactly. NumTombstones no longer exists, so
drop it from the summary.

Written by Claude Opus 4.8
---
 llvm/utils/gdb-scripts/prettyprinters.py | 48 ++++++++++--------------
 llvm/utils/lldbDataFormatters.py         | 39 ++++---------------
 2 files changed, 27 insertions(+), 60 deletions(-)

diff --git a/llvm/utils/gdb-scripts/prettyprinters.py b/llvm/utils/gdb-scripts/prettyprinters.py
index d18944ddac4b2..a97023da58260 100644
--- a/llvm/utils/gdb-scripts/prettyprinters.py
+++ b/llvm/utils/gdb-scripts/prettyprinters.py
@@ -146,43 +146,34 @@ class DenseMapPrinter:
     "Print a DenseMap"
 
     class _iterator:
-        def __init__(self, key_info_t, begin, end):
-            self.key_info_t = key_info_t
-            self.cur = begin
-            self.end = end
+        def __init__(self, buckets, used, num_buckets):
+            self.buckets = buckets
+            self.used = used
+            self.num_buckets = num_buckets
+            self.index = 0
             self.advancePastEmptyBuckets()
             self.first = True
 
         def __iter__(self):
             return self
 
+        def isUsed(self, index):
+            # Occupancy is tracked in a packed 1-bit-per-bucket "used" array of
+            # uint32_t words, so a bucket is occupied iff its bit is set.
+            word = self.used[index >> 5]
+            return (int(word) >> (index & 31)) & 1
+
         def advancePastEmptyBuckets(self):
-            # disabled until the comments below can be addressed
-            # keeping as notes/posterity/hints for future contributors
-            return
-            n = self.key_info_t.name
-            is_equal = gdb.parse_and_eval(n + "::isEqual")
-            empty = gdb.parse_and_eval(n + "::getEmptyKey()")
-            # the following is invalid, GDB fails with:
-            #   Python Exception <class 'gdb.error'> Attempt to take address of value
-            #   not located in memory.
-            # because isEqual took parameter (for the unsigned long key I was testing)
-            # by const ref, and GDB
-            # It's also not entirely general - we should be accessing the "getFirst()"
-            # member function, not the 'first' member variable, but I've yet to figure
-            # out how to find/call member functions (especially (const) overloaded
-            # ones) on a gdb.Value.
-            while self.cur != self.end and
-                is_equal(self.cur.dereference()["first"], empty):
-                self.cur = self.cur + 1
+            while self.index < self.num_buckets and not self.isUsed(self.index):
+                self.index += 1
 
         def __next__(self):
-            if self.cur == self.end:
+            if self.index >= self.num_buckets:
                 raise StopIteration
-            cur = self.cur
-            v = cur.dereference()["first" if self.first else "second"]
+            bucket = (self.buckets + self.index).dereference()
+            v = bucket["first" if self.first else "second"]
             if not self.first:
-                self.cur = self.cur + 1
+                self.index += 1
                 self.advancePastEmptyBuckets()
                 self.first = True
             else:
@@ -197,9 +188,8 @@ def __init__(self, val):
 
     def children(self):
         t = self.val.type.template_argument(3).pointer()
-        begin = self.val["Buckets"].cast(t)
-        end = (begin + self.val["NumBuckets"]).cast(t)
-        return self._iterator(self.val.type.template_argument(2), begin, end)
+        buckets = self.val["Buckets"].cast(t)
+        return self._iterator(buckets, self.val["Used"], int(self.val["NumBuckets"]))
 
     def to_string(self):
         return "llvm::DenseMap with %d elements" % (self.val["NumEntries"])
diff --git a/llvm/utils/lldbDataFormatters.py b/llvm/utils/lldbDataFormatters.py
index c986f6695f88f..785858af64808 100644
--- a/llvm/utils/lldbDataFormatters.py
+++ b/llvm/utils/lldbDataFormatters.py
@@ -6,7 +6,6 @@
 
 from __future__ import annotations
 
-import collections
 from typing import Literal, Optional
 import lldb
 
@@ -448,14 +447,7 @@ def _set_raw_pointer(self, raw_value, min_low_bits):
 def DenseMapSummary(valobj: lldb.SBValue, _) -> str:
     raw_value = valobj.GetNonSyntheticValue()
     num_entries = raw_value.GetChildMemberWithName("NumEntries").unsigned
-    num_tombstones = raw_value.GetChildMemberWithName("NumTombstones").unsigned
-
-    summary = f"size={num_entries}"
-    if num_tombstones == 1:
-        # The heuristic to identify valid entries does not handle the case of a
-        # single tombstone. The summary calls attention to this.
-        summary = f"tombstones=1, {summary}"
-    return summary
+    return f"size={num_entries}"
 
 
 class DenseMapSynthetic:
@@ -492,31 +484,16 @@ def update(self):
         if num_entries == 0:
             return
 
-        buckets = self.valobj.GetChildMemberWithName("Buckets")
         num_buckets = self.valobj.GetChildMemberWithName("NumBuckets").unsigned
+        used = self.valobj.GetChildMemberWithName("Used")
 
-        # Bucket entries contain one of the following:
-        #   1. Valid key-value
-        #   2. Empty key
-        #   3. Tombstone key (a deleted entry)
-        #
-        # NumBuckets is always greater than NumEntries. The empty key, and
-        # potentially the tombstone key, will occur multiple times. A key that
-        # is repeated is either the empty key or the tombstone key.
-
-        # For each key, collect a list of buckets it appears in.
-        key_buckets: dict[str, list[int]] = collections.defaultdict(list)
+        # Occupancy is tracked in a packed 1-bit-per-bucket "used" array of
+        # uint32_t words. A bucket holds a valid entry iff its bit is set;
+        # empty and erased buckets are clear.
         for index in range(num_buckets):
-            bucket = buckets.GetValueForExpressionPath(f"[{index}]")
-            key = bucket.GetChildAtIndex(0)
-            key_buckets[str(key.data)].append(index)
-
-        # Heuristic: This is not a multi-map, any repeated (non-unique) keys are
-        # either the the empty key or the tombstone key. Populate child_buckets
-        # with the indexes of entries containing unique keys.
-        for indexes in key_buckets.values():
-            if len(indexes) == 1:
-                self.child_buckets.append(indexes[0])
+            word = used.GetValueForExpressionPath(f"[{index >> 5}]").unsigned
+            if (word >> (index & 31)) & 1:
+                self.child_buckets.append(index)
 
 
 class DenseSetSynthetic:

>From 24da9995799ab4cd901ffe351bb2dd4fc406ea68 Mon Sep 17 00:00:00 2001
From: Fangrui Song <i at maskray.me>
Date: Sun, 7 Jun 2026 12:05:37 -0700
Subject: [PATCH 2/3] simplify

---
 llvm/utils/lldbDataFormatters.py | 8 +++++---
 1 file changed, 5 insertions(+), 3 deletions(-)

diff --git a/llvm/utils/lldbDataFormatters.py b/llvm/utils/lldbDataFormatters.py
index 785858af64808..e424a2d832449 100644
--- a/llvm/utils/lldbDataFormatters.py
+++ b/llvm/utils/lldbDataFormatters.py
@@ -489,10 +489,12 @@ def update(self):
 
         # Occupancy is tracked in a packed 1-bit-per-bucket "used" array of
         # uint32_t words. A bucket holds a valid entry iff its bit is set;
-        # empty and erased buckets are clear.
+        # empty and erased buckets are clear. Read the whole array in one go
+        # rather than fetching each word with a separate expression path.
+        num_words = (num_buckets + 31) // 32
+        words = used.GetPointeeData(0, num_words).uint32
         for index in range(num_buckets):
-            word = used.GetValueForExpressionPath(f"[{index >> 5}]").unsigned
-            if (word >> (index & 31)) & 1:
+            if (words[index >> 5] >> (index & 31)) & 1:
                 self.child_buckets.append(index)
 
 

>From ae1d28752f0cbe184e7b9ca4d59f4a5380c14b17 Mon Sep 17 00:00:00 2001
From: Fangrui Song <i at maskray.me>
Date: Sun, 7 Jun 2026 19:13:39 -0700
Subject: [PATCH 3/3] Update llvm/utils/lldbDataFormatters.py

Co-authored-by: Dave Lee <davelee.com at gmail.com>
---
 llvm/utils/lldbDataFormatters.py | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/llvm/utils/lldbDataFormatters.py b/llvm/utils/lldbDataFormatters.py
index e424a2d832449..1899f3acbe21b 100644
--- a/llvm/utils/lldbDataFormatters.py
+++ b/llvm/utils/lldbDataFormatters.py
@@ -492,7 +492,7 @@ def update(self):
         # empty and erased buckets are clear. Read the whole array in one go
         # rather than fetching each word with a separate expression path.
         num_words = (num_buckets + 31) // 32
-        words = used.GetPointeeData(0, num_words).uint32
+        words = used.GetPointeeData(0, num_words).uint32s
         for index in range(num_buckets):
             if (words[index >> 5] >> (index & 31)) & 1:
                 self.child_buckets.append(index)



More information about the llvm-commits mailing list