[llvm] e5e3025 - [CAS] Check StandaloneData before reopening a standalone object (#223807)

via llvm-commits llvm-commits at lists.llvm.org
Wed Sep 16 09:04:56 PDT 2026


Author: Steven Wu
Date: 2026-09-16T09:04:50-07:00
New Revision: e5e30255b4cfe0684741b45b16c20ff64e01bca3

URL: https://github.com/llvm/llvm-project/commit/e5e30255b4cfe0684741b45b16c20ff64e01bca3
DIFF: https://github.com/llvm/llvm-project/commit/e5e30255b4cfe0684741b45b16c20ff64e01bca3.diff

LOG: [CAS] Check StandaloneData before reopening a standalone object (#223807)

OnDiskGraphDB::load() inserted every standalone object it mapped into
StandaloneData but never read from it: lookup() and count() had no
callers anywhere. So each load of an object stored outside the data pool
repeated the open, fstat, mmap and close, and then insert() saw the hash
was already present, dropped the freshly created mapping, and returned
the pointer it already had. The whole sequence was wasted.

Consult lookup() before touching the filesystem to make loading already
opened CAS objects much faster.

Added: 
    

Modified: 
    llvm/lib/CAS/OnDiskGraphDB.cpp
    llvm/unittests/CAS/OnDiskGraphDBTest.cpp

Removed: 
    


################################################################################
diff  --git a/llvm/lib/CAS/OnDiskGraphDB.cpp b/llvm/lib/CAS/OnDiskGraphDB.cpp
index a817b35f6f588..7626e0181fd7c 100644
--- a/llvm/lib/CAS/OnDiskGraphDB.cpp
+++ b/llvm/lib/CAS/OnDiskGraphDB.cpp
@@ -1348,6 +1348,11 @@ OnDiskGraphDB::load(ObjectID ExternalRef) {
     break;
   }
 
+  // Search in StandaloneMap to see if data is already loaded.
+  auto *StandaloneMap = static_cast<StandaloneDataMapTy *>(StandaloneData);
+  if (const StandaloneDataInMemory *SDIM = StandaloneMap->lookup(I->Hash))
+    return ObjectHandle::fromMemory(reinterpret_cast<uintptr_t>(SDIM));
+
   // Load it from disk.
   //
   // Note: Creation logic guarantees that data that needs null-termination is
@@ -1376,8 +1381,7 @@ OnDiskGraphDB::load(ObjectID ExternalRef) {
     return createCorruptObjectError(getDigest(*I));
 
   return ObjectHandle::fromMemory(
-      static_cast<StandaloneDataMapTy *>(StandaloneData)
-          ->insert(I->Hash, Object.SK, std::move(Region), I->Offset));
+      StandaloneMap->insert(I->Hash, Object.SK, std::move(Region), I->Offset));
 }
 
 Expected<bool> OnDiskGraphDB::isMaterialized(ObjectID Ref) {

diff  --git a/llvm/unittests/CAS/OnDiskGraphDBTest.cpp b/llvm/unittests/CAS/OnDiskGraphDBTest.cpp
index 02ccd9993621a..83477cb2c12e4 100644
--- a/llvm/unittests/CAS/OnDiskGraphDBTest.cpp
+++ b/llvm/unittests/CAS/OnDiskGraphDBTest.cpp
@@ -8,11 +8,14 @@
 
 #include "CASTestConfig.h"
 #include "OnDiskCommonUtils.h"
+#include "llvm/Support/FileSystem.h"
 #include "llvm/Support/MemoryBuffer.h"
 #include "llvm/Testing/Support/Error.h"
 #include "llvm/Testing/Support/SupportHelpers.h"
 #include "gtest/gtest.h"
 
+#include <set>
+
 using namespace llvm;
 using namespace llvm::cas;
 using namespace llvm::cas::ondisk;
@@ -88,6 +91,51 @@ TEST_F(OnDiskCASTest, OnDiskGraphDBTest) {
   EXPECT_EQ(DB->getStorageSize(), StorageSize);
 }
 
+TEST_F(OnDiskCASTest, OnDiskGraphDBStandaloneObjectMappedOnce) {
+  unittest::TempDir Temp("ondiskcas", /*Unique=*/true);
+  std::unique_ptr<OnDiskGraphDB> DB;
+  ASSERT_THAT_ERROR(
+      OnDiskGraphDB::open(Temp.path(), "blake3", sizeof(HashType)).moveInto(DB),
+      Succeeded());
+
+  auto listFiles = [&Temp]() {
+    std::set<std::string> Files;
+    std::error_code EC;
+    for (sys::fs::directory_iterator I(Temp.path(), EC), E; I != E && !EC;
+         I.increment(EC))
+      Files.insert(I->path());
+    return Files;
+  };
+  std::set<std::string> Before = listFiles();
+
+  // Objects above TrieRecord::MaxEmbeddedSize are written to a file of their
+  // own instead of into the data pool.
+  std::string Data(128 * 1024, 'z');
+  std::optional<ObjectID> ID;
+  ASSERT_THAT_ERROR(store(*DB, Data, {}).moveInto(ID), Succeeded());
+
+  std::optional<ondisk::ObjectHandle> Obj;
+  ASSERT_THAT_ERROR(DB->load(*ID).moveInto(Obj), Succeeded());
+  ASSERT_TRUE(Obj);
+  EXPECT_EQ(toStringRef(DB->getObjectData(*Obj)), Data);
+
+  // Loading it kept the mapping, so deleting the file must not stop a second
+  // load from returning the same bytes. This fails if load() reopens the file
+  // every time instead of consulting the objects it has already mapped.
+  unsigned Removed = 0;
+  for (const std::string &Path : listFiles())
+    if (!Before.count(Path)) {
+      ASSERT_FALSE(sys::fs::remove(Path));
+      ++Removed;
+    }
+  ASSERT_GE(Removed, 1u) << "expected a standalone file to have been created";
+
+  std::optional<ondisk::ObjectHandle> Reloaded;
+  ASSERT_THAT_ERROR(DB->load(*ID).moveInto(Reloaded), Succeeded());
+  ASSERT_TRUE(Reloaded);
+  EXPECT_EQ(toStringRef(DB->getObjectData(*Reloaded)), Data);
+}
+
 TEST_F(OnDiskCASTest, OnDiskGraphDBFaultInSingleNode) {
   unittest::TempDir TempUpstream("ondiskcas-upstream", /*Unique=*/true);
   std::unique_ptr<OnDiskGraphDB> UpstreamDB;


        


More information about the llvm-commits mailing list