[llvm] [llvm-strings] Use small buffer instead of reading whole file (PR #163073)

via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 25 13:17:17 PDT 2026


https://github.com/aokblast updated https://github.com/llvm/llvm-project/pull/163073

>From 42984b543dafa8731fbf2034331c8d81c8d7ed65 Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Sun, 12 Oct 2025 22:01:13 +0800
Subject: [PATCH 01/20] [llvm-strings] Use small buffer instead of reading
 whole file

To prevent large memory footprint, we use small buffer and a seperated
string to print. This can largely reduce the memory consumption when
processing.

In the following example, the test.file is 1G regular file, which is possible in the
daily workload.

cat ./test.file | llvm-strings --radix=o > out &&
ps -o command,vsz,rss | grep strings

Reading whole file:
llvm-strings 5129824 3999920

Small buffer implementation:
llvm-strings 17604   5476

Also, it does not affect too much runtime:

time llvm-strings --radix=o > out < ./test.file

Reading whole file:
________________________________________________________
Executed in    5.67 secs    fish           external
   usr time    4.38 secs    2.40 millis    4.38 secs
   sys time    1.28 secs    0.00 millis    1.28 secs

Small buffer implementation:
________________________________________________________
Executed in    5.87 secs    fish           external
   usr time    5.65 secs    2.59 millis    5.65 secs
   sys time    0.21 secs    0.00 millis    0.21 secs
---
 llvm/tools/llvm-strings/llvm-strings.cpp | 74 ++++++++++++++++++------
 1 file changed, 55 insertions(+), 19 deletions(-)

diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 9979b93de84270..425c9d9831b99c 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -18,7 +18,9 @@
 #include "llvm/Option/ArgList.h"
 #include "llvm/Option/Option.h"
 #include "llvm/Support/CommandLine.h"
+#include "llvm/Support/Errno.h"
 #include "llvm/Support/Error.h"
+#include "llvm/Support/FileSystem.h"
 #include "llvm/Support/Format.h"
 #include "llvm/Support/InitLLVM.h"
 #include "llvm/Support/MemoryBuffer.h"
@@ -88,7 +90,9 @@ static void parseIntArg(const opt::InputArgList &Args, int ID, T &Value) {
   }
 }
 
-static void strings(raw_ostream &OS, StringRef FileName, StringRef Contents) {
+static void strings(raw_ostream &OS, StringRef FileName,
+                    sys::fs::file_t FileHandle) {
+  SmallString<sys::fs::DefaultReadChunkSize> Buffer;
   auto print = [&OS, FileName](unsigned Offset, StringRef L) {
     if (L.size() < static_cast<size_t>(MinLength))
       return;
@@ -110,19 +114,42 @@ static void strings(raw_ostream &OS, StringRef FileName, StringRef Contents) {
     OS << L << '\n';
   };
 
-  const char *B = Contents.begin();
-  const char *P = nullptr, *E = nullptr, *S = nullptr;
-  for (P = Contents.begin(), E = Contents.end(); P < E; ++P) {
-    if (isPrint(*P) || *P == '\t') {
-      if (S == nullptr)
-        S = P;
-    } else if (S) {
-      print(S - B, StringRef(S, P - S));
-      S = nullptr;
+  unsigned Offset = 0, LocalOffset = 0, CurSize = 0;
+  Buffer.resize(sys::fs::DefaultReadChunkSize);
+  auto FillBuffer = [&Buffer, FileHandle, &FileName]() -> unsigned {
+    Expected<size_t> ReadBytesOrErr = sys::fs::readNativeFile(
+        FileHandle,
+        MutableArrayRef(Buffer.begin(), sys::fs::DefaultReadChunkSize));
+    if (!ReadBytesOrErr) {
+      errs() << FileName << ": "
+             << errorToErrorCode(ReadBytesOrErr.takeError()).message() << '\n';
+      return 0;
+    }
+    return *ReadBytesOrErr;
+  };
+  std::string StringBuffer;
+  while (true) {
+    if (LocalOffset == CurSize) {
+      CurSize = FillBuffer();
+      if (CurSize == 0)
+        break;
+      LocalOffset = 0;
+    }
+    char C = Buffer[LocalOffset++];
+    if (isPrint(C) || C == '\t') {
+      StringBuffer.push_back(C);
+    } else if (StringBuffer.size()) {
+      print(Offset, StringRef(StringBuffer.c_str(), StringBuffer.size()));
+      Offset += StringBuffer.size();
+      StringBuffer.clear();
+      ++Offset;
+    } else {
+      ++Offset;
     }
   }
-  if (S)
-    print(S - B, StringRef(S, E - S));
+
+  if (StringBuffer.size())
+    print(Offset, StringRef(StringBuffer.c_str(), StringBuffer.size()));
 }
 
 int main(int argc, char **argv) {
@@ -174,13 +201,22 @@ int main(int argc, char **argv) {
     InputFileNames.push_back("-");
 
   for (const auto &File : InputFileNames) {
-    ErrorOr<std::unique_ptr<MemoryBuffer>> Buffer =
-        MemoryBuffer::getFileOrSTDIN(File, /*IsText=*/true);
-    if (std::error_code EC = Buffer.getError())
-      errs() << File << ": " << EC.message() << '\n';
-    else
-      strings(llvm::outs(), File == "-" ? "{standard input}" : File,
-              Buffer.get()->getMemBufferRef().getBuffer());
+    sys::fs::file_t FD;
+    if (File == "-") {
+      FD = sys::fs::getStdinHandle();
+    } else {
+      Expected<sys::fs::file_t> FDOrErr =
+          sys::fs::openNativeFileForRead(File, sys::fs::OF_None);
+      ;
+      if (!FDOrErr) {
+        errs() << File << ": "
+               << errorToErrorCode(FDOrErr.takeError()).message() << '\n';
+        continue;
+      }
+      FD = *FDOrErr;
+    }
+
+    strings(llvm::outs(), File == "-" ? "{standard input}" : File, FD);
   }
 
   return EXIT_SUCCESS;

>From f2a884ed210f032ed9288b5de18d6dd211afbc15 Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Sun, 26 Jul 2026 00:51:10 +0800
Subject: [PATCH 02/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 llvm/tools/llvm-strings/llvm-strings.cpp | 70 ++++++++++--------------
 1 file changed, 28 insertions(+), 42 deletions(-)

diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 425c9d9831b99c..888e8d8bdf955f 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -27,7 +27,10 @@
 #include "llvm/Support/Program.h"
 #include "llvm/Support/WithColor.h"
 #include <cctype>
+#include <fstream>
+#include <iostream>
 #include <string>
+#include <vector>
 
 using namespace llvm;
 using namespace llvm::object;
@@ -90,8 +93,7 @@ static void parseIntArg(const opt::InputArgList &Args, int ID, T &Value) {
   }
 }
 
-static void strings(raw_ostream &OS, StringRef FileName,
-                    sys::fs::file_t FileHandle) {
+static void strings(raw_ostream &OS, StringRef FileName, std::istream &IS) {
   SmallString<sys::fs::DefaultReadChunkSize> Buffer;
   auto print = [&OS, FileName](unsigned Offset, StringRef L) {
     if (L.size() < static_cast<size_t>(MinLength))
@@ -114,38 +116,28 @@ static void strings(raw_ostream &OS, StringRef FileName,
     OS << L << '\n';
   };
 
-  unsigned Offset = 0, LocalOffset = 0, CurSize = 0;
-  Buffer.resize(sys::fs::DefaultReadChunkSize);
-  auto FillBuffer = [&Buffer, FileHandle, &FileName]() -> unsigned {
-    Expected<size_t> ReadBytesOrErr = sys::fs::readNativeFile(
-        FileHandle,
-        MutableArrayRef(Buffer.begin(), sys::fs::DefaultReadChunkSize));
-    if (!ReadBytesOrErr) {
-      errs() << FileName << ": "
-             << errorToErrorCode(ReadBytesOrErr.takeError()).message() << '\n';
-      return 0;
-    }
-    return *ReadBytesOrErr;
-  };
   std::string StringBuffer;
+  unsigned Offset = 0;
   while (true) {
-    if (LocalOffset == CurSize) {
-      CurSize = FillBuffer();
-      if (CurSize == 0)
-        break;
-      LocalOffset = 0;
-    }
-    char C = Buffer[LocalOffset++];
-    if (isPrint(C) || C == '\t') {
-      StringBuffer.push_back(C);
-    } else if (StringBuffer.size()) {
-      print(Offset, StringRef(StringBuffer.c_str(), StringBuffer.size()));
-      Offset += StringBuffer.size();
-      StringBuffer.clear();
-      ++Offset;
-    } else {
-      ++Offset;
+    IS.read(Buffer.data(), Buffer.size());
+    std::streamsize CurSize = IS.gcount();
+    if (CurSize <= 0)
+      break;
+    for (std::streamsize I = 0; I != CurSize; ++I) {
+      char C = Buffer[I];
+      if (isPrint(C) || C == '\t') {
+        StringBuffer.push_back(C);
+      } else if (StringBuffer.size()) {
+        print(Offset, StringRef(StringBuffer.c_str(), StringBuffer.size()));
+        Offset += StringBuffer.size();
+        StringBuffer.clear();
+        ++Offset;
+      } else {
+        ++Offset;
+      }
     }
+    if (!IS)
+      break;
   }
 
   if (StringBuffer.size())
@@ -201,22 +193,16 @@ int main(int argc, char **argv) {
     InputFileNames.push_back("-");
 
   for (const auto &File : InputFileNames) {
-    sys::fs::file_t FD;
     if (File == "-") {
-      FD = sys::fs::getStdinHandle();
+      strings(llvm::outs(), "{standard input}", std::cin);
     } else {
-      Expected<sys::fs::file_t> FDOrErr =
-          sys::fs::openNativeFileForRead(File, sys::fs::OF_None);
-      ;
-      if (!FDOrErr) {
-        errs() << File << ": "
-               << errorToErrorCode(FDOrErr.takeError()).message() << '\n';
+      std::ifstream IS(File, std::ios::in | std::ios::binary);
+      if (!IS) {
+        errs() << File << ": " << sys::StrError(errno) << '\n';
         continue;
       }
-      FD = *FDOrErr;
+      strings(llvm::outs(), File, IS);
     }
-
-    strings(llvm::outs(), File == "-" ? "{standard input}" : File, FD);
   }
 
   return EXIT_SUCCESS;

>From d63e61523bd40d2cd116c1efde793f1d16abfc20 Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Sun, 26 Jul 2026 01:37:51 +0800
Subject: [PATCH 03/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 llvm/tools/llvm-strings/llvm-strings.cpp | 1 +
 1 file changed, 1 insertion(+)

diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 888e8d8bdf955f..2dd1450eb50e9d 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -95,6 +95,7 @@ static void parseIntArg(const opt::InputArgList &Args, int ID, T &Value) {
 
 static void strings(raw_ostream &OS, StringRef FileName, std::istream &IS) {
   SmallString<sys::fs::DefaultReadChunkSize> Buffer;
+  Buffer.resize_for_overwrite(sys::fs::DefaultReadChunkSize);
   auto print = [&OS, FileName](unsigned Offset, StringRef L) {
     if (L.size() < static_cast<size_t>(MinLength))
       return;

>From 20b9d55d766aac46f0ecbed287d8584c73491156 Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Mon, 27 Jul 2026 19:27:58 +0800
Subject: [PATCH 04/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 llvm/tools/llvm-strings/llvm-strings.cpp | 33 ++++++++++++++----------
 1 file changed, 19 insertions(+), 14 deletions(-)

diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 2dd1450eb50e9d..58973797d3fb3b 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -12,6 +12,7 @@
 //===----------------------------------------------------------------------===//
 
 #include "Opts.inc"
+#include "llvm/ADT/ArrayRef.h"
 #include "llvm/ADT/StringExtras.h"
 #include "llvm/Object/Binary.h"
 #include "llvm/Option/Arg.h"
@@ -28,9 +29,7 @@
 #include "llvm/Support/WithColor.h"
 #include <cctype>
 #include <fstream>
-#include <iostream>
 #include <string>
-#include <vector>
 
 using namespace llvm;
 using namespace llvm::object;
@@ -93,9 +92,9 @@ static void parseIntArg(const opt::InputArgList &Args, int ID, T &Value) {
   }
 }
 
-static void strings(raw_ostream &OS, StringRef FileName, std::istream &IS) {
+static void strings(raw_ostream &OS, StringRef FileName,
+                    sys::fs::file_t Handle) {
   SmallString<sys::fs::DefaultReadChunkSize> Buffer;
-  Buffer.resize_for_overwrite(sys::fs::DefaultReadChunkSize);
   auto print = [&OS, FileName](unsigned Offset, StringRef L) {
     if (L.size() < static_cast<size_t>(MinLength))
       return;
@@ -117,14 +116,21 @@ static void strings(raw_ostream &OS, StringRef FileName, std::istream &IS) {
     OS << L << '\n';
   };
 
+  Buffer.resize_for_overwrite(sys::fs::DefaultReadChunkSize);
   std::string StringBuffer;
   unsigned Offset = 0;
   while (true) {
-    IS.read(Buffer.data(), Buffer.size());
-    std::streamsize CurSize = IS.gcount();
+    Expected<size_t> ReadBytesOrErr = sys::fs::readNativeFile(
+        Handle, MutableArrayRef(Buffer.data(), Buffer.size()));
+    if (!ReadBytesOrErr) {
+      errs() << FileName << ": "
+             << errorToErrorCode(ReadBytesOrErr.takeError()).message() << '\n';
+      return;
+    }
+    std::size_t CurSize = *ReadBytesOrErr;
     if (CurSize <= 0)
       break;
-    for (std::streamsize I = 0; I != CurSize; ++I) {
+    for (std::size_t I = 0; I != CurSize; ++I) {
       char C = Buffer[I];
       if (isPrint(C) || C == '\t') {
         StringBuffer.push_back(C);
@@ -137,8 +143,6 @@ static void strings(raw_ostream &OS, StringRef FileName, std::istream &IS) {
         ++Offset;
       }
     }
-    if (!IS)
-      break;
   }
 
   if (StringBuffer.size())
@@ -195,14 +199,15 @@ int main(int argc, char **argv) {
 
   for (const auto &File : InputFileNames) {
     if (File == "-") {
-      strings(llvm::outs(), "{standard input}", std::cin);
+      strings(llvm::outs(), "{standard input}", sys::fs::getStdinHandle());
     } else {
-      std::ifstream IS(File, std::ios::in | std::ios::binary);
-      if (!IS) {
-        errs() << File << ": " << sys::StrError(errno) << '\n';
+      Expected<sys::fs::file_t> FDOrErr = sys::fs::openNativeFileForReadWrite(
+          File, sys::fs::CD_OpenExisting, sys::fs::OF_None);
+      if (!FDOrErr) {
+        errs() << File << ": " << FDOrErr.takeError() << '\n';
         continue;
       }
-      strings(llvm::outs(), File, IS);
+      strings(llvm::outs(), File, *FDOrErr);
     }
   }
 

>From a539beb754b4b9c8ba386cfae54715be93ce59cc Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Mon, 27 Jul 2026 19:33:52 +0800
Subject: [PATCH 05/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 llvm/tools/llvm-strings/llvm-strings.cpp | 3 ---
 1 file changed, 3 deletions(-)

diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 58973797d3fb3b..763872281349d1 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -12,14 +12,12 @@
 //===----------------------------------------------------------------------===//
 
 #include "Opts.inc"
-#include "llvm/ADT/ArrayRef.h"
 #include "llvm/ADT/StringExtras.h"
 #include "llvm/Object/Binary.h"
 #include "llvm/Option/Arg.h"
 #include "llvm/Option/ArgList.h"
 #include "llvm/Option/Option.h"
 #include "llvm/Support/CommandLine.h"
-#include "llvm/Support/Errno.h"
 #include "llvm/Support/Error.h"
 #include "llvm/Support/FileSystem.h"
 #include "llvm/Support/Format.h"
@@ -28,7 +26,6 @@
 #include "llvm/Support/Program.h"
 #include "llvm/Support/WithColor.h"
 #include <cctype>
-#include <fstream>
 #include <string>
 
 using namespace llvm;

>From 2d49b81bad90667262c72d5a5095ce756fecf3ab Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Thu, 13 Aug 2026 14:32:07 +0800
Subject: [PATCH 06/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 llvm/tools/llvm-strings/llvm-strings.cpp | 84 ++++++++++++++----------
 1 file changed, 48 insertions(+), 36 deletions(-)

diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 763872281349d1..2ce9b913dac7fa 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -69,7 +69,8 @@ static StringRef ToolName;
 static cl::list<std::string> InputFileNames(cl::Positional,
                                             cl::desc("<input object files>"));
 
-static int MinLength = 4;
+static constexpr int DefaultMinLength = 4;
+static int MinLength = DefaultMinLength;
 static bool PrintFileName;
 
 enum radix { none, octal, hexadecimal, decimal };
@@ -92,31 +93,39 @@ static void parseIntArg(const opt::InputArgList &Args, int ID, T &Value) {
 static void strings(raw_ostream &OS, StringRef FileName,
                     sys::fs::file_t Handle) {
   SmallString<sys::fs::DefaultReadChunkSize> Buffer;
-  auto print = [&OS, FileName](unsigned Offset, StringRef L) {
-    if (L.size() < static_cast<size_t>(MinLength))
-      return;
-    if (PrintFileName)
-      OS << FileName << ": ";
-    switch (Radix) {
-    case none:
-      break;
-    case octal:
-      OS << format("%7o ", Offset);
-      break;
-    case hexadecimal:
-      OS << format("%7x ", Offset);
-      break;
-    case decimal:
-      OS << format("%7u ", Offset);
-      break;
+  SmallString<DefaultMinLength> Prefix;
+  auto print = [&OS, FileName, &Prefix](unsigned Offset, StringRef L) {
+    if (Prefix.size() + L.size() >= static_cast<size_t>(MinLength)) {
+      if (PrintFileName)
+        OS << FileName << ": ";
+      switch (Radix) {
+      case none:
+        break;
+      case octal:
+        OS << format("%7o ", Offset);
+        break;
+      case hexadecimal:
+        OS << format("%7x ", Offset);
+        break;
+      case decimal:
+        OS << format("%7u ", Offset);
+        break;
+      }
+      OS << Prefix << L << '\n';
     }
-    OS << L << '\n';
+    Prefix.clear();
   };
 
   Buffer.resize_for_overwrite(sys::fs::DefaultReadChunkSize);
-  std::string StringBuffer;
+  // Offset of the start of the current chunk within the file.
   unsigned Offset = 0;
   while (true) {
+    /*
+     * llvm-strings should be able to process a very large file on a
+     * memory-budgeted machine. To handle this, we read the file in chunk
+     * instead of allocate a very large memory and copy the whole file to the
+     * memory.
+     */
     Expected<size_t> ReadBytesOrErr = sys::fs::readNativeFile(
         Handle, MutableArrayRef(Buffer.data(), Buffer.size()));
     if (!ReadBytesOrErr) {
@@ -125,25 +134,28 @@ static void strings(raw_ostream &OS, StringRef FileName,
       return;
     }
     std::size_t CurSize = *ReadBytesOrErr;
-    if (CurSize <= 0)
+    if (CurSize == 0)
       break;
-    for (std::size_t I = 0; I != CurSize; ++I) {
-      char C = Buffer[I];
-      if (isPrint(C) || C == '\t') {
-        StringBuffer.push_back(C);
-      } else if (StringBuffer.size()) {
-        print(Offset, StringRef(StringBuffer.c_str(), StringBuffer.size()));
-        Offset += StringBuffer.size();
-        StringBuffer.clear();
-        ++Offset;
-      } else {
-        ++Offset;
+
+    const char *B = Buffer.data();
+    const char *E = B + CurSize;
+    const char *S = Prefix.empty() ? nullptr : B;
+    for (const char *P = B; P != E; ++P) {
+      if (isPrint(*P) || *P == '\t') {
+        if (!S)
+          S = P;
+      } else if (S) {
+        print(Offset + (S - B) - Prefix.size(), StringRef(S, P - S));
+        S = nullptr;
       }
     }
+    if (S)
+      Prefix.append(S, E);
+    Offset += CurSize;
   }
 
-  if (StringBuffer.size())
-    print(Offset, StringRef(StringBuffer.c_str(), StringBuffer.size()));
+  if (!Prefix.empty())
+    print(Offset - Prefix.size(), StringRef());
 }
 
 int main(int argc, char **argv) {
@@ -198,8 +210,8 @@ int main(int argc, char **argv) {
     if (File == "-") {
       strings(llvm::outs(), "{standard input}", sys::fs::getStdinHandle());
     } else {
-      Expected<sys::fs::file_t> FDOrErr = sys::fs::openNativeFileForReadWrite(
-          File, sys::fs::CD_OpenExisting, sys::fs::OF_None);
+      Expected<sys::fs::file_t> FDOrErr =
+          sys::fs::openNativeFileForRead(File, sys::fs::OF_TextWithCRLF);
       if (!FDOrErr) {
         errs() << File << ": " << FDOrErr.takeError() << '\n';
         continue;

>From 83b62db4b0c774bf5047d4684248ccfda658e019 Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Mon, 17 Aug 2026 03:01:07 -0700
Subject: [PATCH 07/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 .../tools/llvm-strings/chunk-boundary.test    | 56 +++++++++++
 llvm/test/tools/llvm-strings/errors.test      |  9 ++
 llvm/test/tools/llvm-strings/read-error.test  | 13 +++
 llvm/tools/llvm-strings/llvm-strings.cpp      | 99 ++++++++++---------
 4 files changed, 133 insertions(+), 44 deletions(-)
 create mode 100644 llvm/test/tools/llvm-strings/chunk-boundary.test
 create mode 100644 llvm/test/tools/llvm-strings/errors.test
 create mode 100644 llvm/test/tools/llvm-strings/read-error.test

diff --git a/llvm/test/tools/llvm-strings/chunk-boundary.test b/llvm/test/tools/llvm-strings/chunk-boundary.test
new file mode 100644
index 00000000000000..9949bd76e51161
--- /dev/null
+++ b/llvm/test/tools/llvm-strings/chunk-boundary.test
@@ -0,0 +1,56 @@
+## Show that strings interacting with the read-chunk boundary are reported
+## correctly. The input files are crafted assuming the native read chunk size
+## (sys::fs::DefaultReadChunkSize) of 16384 bytes.
+
+## Case 1: at least min string size appears before the boundary, unprintable
+## byte as first byte of the next chunk. The string is printed on its own,
+## with the offset of its start (0x3ff8 = 16384 - 8).
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16376 + b'ENDCHUNK' + b'\0' * 4)" > %t.1
+RUN: llvm-strings --radix=x %t.1 | FileCheck %s --check-prefix=CASE1 --strict-whitespace --implicit-check-not={{.}}
+
+CASE1:{{^}}   3ff8 ENDCHUNK{{$}}
+
+## Case 2: at least min string size appears before the boundary, printable
+## byte as first byte of the next chunk. The prefix is printed together with
+## the following characters, as a single string.
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16378 + b'BEFORE' + b'AFTER!' + b'\0' * 4)" > %t.2
+RUN: llvm-strings --radix=x %t.2 | FileCheck %s --check-prefix=CASE2 --strict-whitespace --implicit-check-not={{.}}
+
+CASE2:{{^}}   3ffa BEFOREAFTER!{{$}}
+
+## Case 3: less than min string size appears before the boundary, unprintable
+## byte as the next byte. The prefix is not printed.
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16382 + b'AB' + b'\0' * 4)" > %t.3
+RUN: llvm-strings %t.3 | count 0
+
+## Case 4: less than min string size appears before the boundary, printable
+## bytes as the next bytes, forming a min length string. The prefix is
+## printed together with the following characters, with the offset of its
+## true start (0x3ffd = 16384 - 3).
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16381 + b'ABC' + b'DEF' + b'\0' * 4)" > %t.4
+RUN: llvm-strings --radix=x %t.4 | FileCheck %s --check-prefix=CASE4 --strict-whitespace --implicit-check-not={{.}}
+
+CASE4:{{^}}   3ffd ABCDEF{{$}}
+
+## Case 5: the prefix is empty at the start of a chunk that starts with a min
+## length string (0x4000 = 16384).
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16384 + b'FRESH' + b'\0' * 4)" > %t.5
+RUN: llvm-strings --radix=x %t.5 | FileCheck %s --check-prefix=CASE5 --strict-whitespace --implicit-check-not={{.}}
+
+CASE5:{{^}}   4000 FRESH{{$}}
+
+## Case 6: a string that spans the entirety of one chunk, with one character
+## before the chunk start and one after its end. The string is printed
+## intact, once. The expected output is generated rather than written as a
+## CHECK line because it is a chunk long.
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16383 + b'S' + b'M' * 16384 + b'E' + b'\0' * 4)" > %t.6
+RUN: %python -c "import sys; sys.stdout.write('S' + 'M' * 16384 + 'E\n')" > %t.6.expected
+RUN: llvm-strings %t.6 > %t.6.out
+RUN: diff %t.6.expected %t.6.out
+
+## A string terminated by the end of the file (no trailing unprintable byte)
+## must still be printed.
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16380 + b'TRAILING')" > %t.7
+RUN: llvm-strings %t.7 | FileCheck %s --check-prefix=EOF --strict-whitespace --implicit-check-not={{.}}
+
+EOF:{{^}}TRAILING{{$}}
diff --git a/llvm/test/tools/llvm-strings/errors.test b/llvm/test/tools/llvm-strings/errors.test
new file mode 100644
index 00000000000000..d2433665bb57a1
--- /dev/null
+++ b/llvm/test/tools/llvm-strings/errors.test
@@ -0,0 +1,9 @@
+## Show that a file that cannot be opened is reported on stderr, and that
+## processing continues with the remaining inputs.
+
+RUN: rm -rf %t && mkdir -p %t
+RUN: echo abcd > %t/good
+RUN: llvm-strings %t/does-not-exist %t/good 2>&1 | FileCheck %s -DFILE=%t/does-not-exist
+
+CHECK: [[FILE]]: {{[Nn]}}o such file or directory
+CHECK: abcd
diff --git a/llvm/test/tools/llvm-strings/read-error.test b/llvm/test/tools/llvm-strings/read-error.test
new file mode 100644
index 00000000000000..75e49399655e01
--- /dev/null
+++ b/llvm/test/tools/llvm-strings/read-error.test
@@ -0,0 +1,13 @@
+## Show that a file that opens but cannot be read is reported on stderr, and
+## that processing continues with the remaining inputs. A directory can be
+## opened for reading on POSIX systems, but reading from it fails with
+## EISDIR. On Windows opening the directory fails instead, so the read error
+## path is not reachable this way.
+# UNSUPPORTED: system-windows
+
+RUN: rm -rf %t && mkdir -p %t/dir
+RUN: echo abcd > %t/good
+RUN: llvm-strings %t/dir %t/good 2>&1 | FileCheck %s -DFILE=%t/dir
+
+CHECK: [[FILE]]: {{[Ii]}}s a directory
+CHECK: abcd
diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 2ce9b913dac7fa..b575601c9aca1d 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -90,42 +90,38 @@ static void parseIntArg(const opt::InputArgList &Args, int ID, T &Value) {
   }
 }
 
+static bool isStringChar(char C) { return isPrint(C) || C == '\t'; }
+
 static void strings(raw_ostream &OS, StringRef FileName,
                     sys::fs::file_t Handle) {
   SmallString<sys::fs::DefaultReadChunkSize> Buffer;
-  SmallString<DefaultMinLength> Prefix;
-  auto print = [&OS, FileName, &Prefix](unsigned Offset, StringRef L) {
-    if (Prefix.size() + L.size() >= static_cast<size_t>(MinLength)) {
-      if (PrintFileName)
-        OS << FileName << ": ";
-      switch (Radix) {
-      case none:
-        break;
-      case octal:
-        OS << format("%7o ", Offset);
-        break;
-      case hexadecimal:
-        OS << format("%7x ", Offset);
-        break;
-      case decimal:
-        OS << format("%7u ", Offset);
-        break;
-      }
-      OS << Prefix << L << '\n';
+  auto printHeader = [&OS, FileName](unsigned StringStart) {
+    if (PrintFileName)
+      OS << FileName << ": ";
+    switch (Radix) {
+    case none:
+      break;
+    case octal:
+      OS << format("%7o ", StringStart);
+      break;
+    case hexadecimal:
+      OS << format("%7x ", StringStart);
+      break;
+    case decimal:
+      OS << format("%7u ", StringStart);
+      break;
     }
-    Prefix.clear();
   };
 
+  // llvm-strings should be able to process a very large file on a
+  // memory-budgeted machine, so the file is read in chunks. To handle this, we
+  // read the file in chunk instead of copying the whole file into memory.
+  SmallString<DefaultMinLength> Candidate;
+  bool InString = false;
+  unsigned StringStart = 0, Offset = 0;
+
   Buffer.resize_for_overwrite(sys::fs::DefaultReadChunkSize);
-  // Offset of the start of the current chunk within the file.
-  unsigned Offset = 0;
   while (true) {
-    /*
-     * llvm-strings should be able to process a very large file on a
-     * memory-budgeted machine. To handle this, we read the file in chunk
-     * instead of allocate a very large memory and copy the whole file to the
-     * memory.
-     */
     Expected<size_t> ReadBytesOrErr = sys::fs::readNativeFile(
         Handle, MutableArrayRef(Buffer.data(), Buffer.size()));
     if (!ReadBytesOrErr) {
@@ -137,25 +133,40 @@ static void strings(raw_ostream &OS, StringRef FileName,
     if (CurSize == 0)
       break;
 
-    const char *B = Buffer.data();
-    const char *E = B + CurSize;
-    const char *S = Prefix.empty() ? nullptr : B;
-    for (const char *P = B; P != E; ++P) {
-      if (isPrint(*P) || *P == '\t') {
-        if (!S)
-          S = P;
-      } else if (S) {
-        print(Offset + (S - B) - Prefix.size(), StringRef(S, P - S));
-        S = nullptr;
+    std::size_t I = 0;
+    while (I != CurSize) {
+      if (InString) {
+        std::size_t Start = I;
+        while (I != CurSize && isStringChar(Buffer[I]))
+          ++I;
+        OS << StringRef(Buffer.data() + Start, I - Start);
+        Offset += I - Start;
+        if (I != CurSize) {
+          OS << '\n';
+          InString = false;
+        }
+      } else if (isStringChar(Buffer[I])) {
+        if (Candidate.empty())
+          StringStart = Offset;
+        Candidate.push_back(Buffer[I]);
+        ++I;
+        ++Offset;
+        if (Candidate.size() >= static_cast<size_t>(MinLength)) {
+          printHeader(StringStart);
+          OS << Candidate;
+          Candidate.clear();
+          InString = true;
+        }
+      } else {
+        Candidate.clear();
+        ++I;
+        ++Offset;
       }
     }
-    if (S)
-      Prefix.append(S, E);
-    Offset += CurSize;
   }
 
-  if (!Prefix.empty())
-    print(Offset - Prefix.size(), StringRef());
+  if (InString)
+    OS << '\n';
 }
 
 int main(int argc, char **argv) {
@@ -213,7 +224,7 @@ int main(int argc, char **argv) {
       Expected<sys::fs::file_t> FDOrErr =
           sys::fs::openNativeFileForRead(File, sys::fs::OF_TextWithCRLF);
       if (!FDOrErr) {
-        errs() << File << ": " << FDOrErr.takeError() << '\n';
+        errs() << File << ": " << toString(FDOrErr.takeError()) << '\n';
         continue;
       }
       strings(llvm::outs(), File, *FDOrErr);

>From 23be9bf15e3310ed0a38e2e55fb6b5a9c388a15e Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Sun, 23 Aug 2026 04:24:22 +0800
Subject: [PATCH 08/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 llvm/test/tools/llvm-strings/errors.test     |  9 --
 llvm/test/tools/llvm-strings/read-error.test | 13 ---
 llvm/tools/llvm-strings/llvm-strings.cpp     | 99 +++++++++++++-------
 3 files changed, 67 insertions(+), 54 deletions(-)
 delete mode 100644 llvm/test/tools/llvm-strings/errors.test
 delete mode 100644 llvm/test/tools/llvm-strings/read-error.test

diff --git a/llvm/test/tools/llvm-strings/errors.test b/llvm/test/tools/llvm-strings/errors.test
deleted file mode 100644
index d2433665bb57a1..00000000000000
--- a/llvm/test/tools/llvm-strings/errors.test
+++ /dev/null
@@ -1,9 +0,0 @@
-## Show that a file that cannot be opened is reported on stderr, and that
-## processing continues with the remaining inputs.
-
-RUN: rm -rf %t && mkdir -p %t
-RUN: echo abcd > %t/good
-RUN: llvm-strings %t/does-not-exist %t/good 2>&1 | FileCheck %s -DFILE=%t/does-not-exist
-
-CHECK: [[FILE]]: {{[Nn]}}o such file or directory
-CHECK: abcd
diff --git a/llvm/test/tools/llvm-strings/read-error.test b/llvm/test/tools/llvm-strings/read-error.test
deleted file mode 100644
index 75e49399655e01..00000000000000
--- a/llvm/test/tools/llvm-strings/read-error.test
+++ /dev/null
@@ -1,13 +0,0 @@
-## Show that a file that opens but cannot be read is reported on stderr, and
-## that processing continues with the remaining inputs. A directory can be
-## opened for reading on POSIX systems, but reading from it fails with
-## EISDIR. On Windows opening the directory fails instead, so the read error
-## path is not reachable this way.
-# UNSUPPORTED: system-windows
-
-RUN: rm -rf %t && mkdir -p %t/dir
-RUN: echo abcd > %t/good
-RUN: llvm-strings %t/dir %t/good 2>&1 | FileCheck %s -DFILE=%t/dir
-
-CHECK: [[FILE]]: {{[Ii]}}s a directory
-CHECK: abcd
diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index b575601c9aca1d..c810ee090373ff 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -113,12 +113,20 @@ static void strings(raw_ostream &OS, StringRef FileName,
     }
   };
 
-  // llvm-strings should be able to process a very large file on a
-  // memory-budgeted machine, so the file is read in chunks. To handle this, we
-  // read the file in chunk instead of copying the whole file into memory.
+  // To handle very large files without consuming excessive memory, we read the
+  // file in a little at a time and process it then rather than reading the
+  // entire file at once.
+  //
+  // A string is only buffered until it is known to be long enough to print;
+  // from then on it is streamed out directly, so an arbitrarily long string
+  // never needs an arbitrarily large buffer. Candidate therefore only ever
+  // holds a run that is shorter than MinLength and that was cut off by the end
+  // of a chunk.
+  const size_t Min = MinLength;
   SmallString<DefaultMinLength> Candidate;
   bool InString = false;
-  unsigned StringStart = 0, Offset = 0;
+  // Offset of the start of the current chunk within the file.
+  unsigned ChunkOffset = 0;
 
   Buffer.resize_for_overwrite(sys::fs::DefaultReadChunkSize);
   while (true) {
@@ -129,40 +137,65 @@ static void strings(raw_ostream &OS, StringRef FileName,
              << errorToErrorCode(ReadBytesOrErr.takeError()).message() << '\n';
       return;
     }
-    std::size_t CurSize = *ReadBytesOrErr;
-    if (CurSize == 0)
+    std::size_t ChunkSize = *ReadBytesOrErr;
+    if (ChunkSize == 0)
       break;
 
-    std::size_t I = 0;
-    while (I != CurSize) {
+    const char *const B = Buffer.data();
+    const char *const E = B + ChunkSize;
+    const char *P = B;
+
+    if (InString || !Candidate.empty()) {
+      while (P != E && isStringChar(*P))
+        ++P;
+      size_t Len = P - B;
       if (InString) {
-        std::size_t Start = I;
-        while (I != CurSize && isStringChar(Buffer[I]))
-          ++I;
-        OS << StringRef(Buffer.data() + Start, I - Start);
-        Offset += I - Start;
-        if (I != CurSize) {
-          OS << '\n';
-          InString = false;
-        }
-      } else if (isStringChar(Buffer[I])) {
-        if (Candidate.empty())
-          StringStart = Offset;
-        Candidate.push_back(Buffer[I]);
-        ++I;
-        ++Offset;
-        if (Candidate.size() >= static_cast<size_t>(MinLength)) {
-          printHeader(StringStart);
-          OS << Candidate;
-          Candidate.clear();
-          InString = true;
-        }
+        OS << StringRef(B, Len);
+      } else if (Candidate.size() + Len >= Min) {
+        printHeader(ChunkOffset - Candidate.size());
+        OS << Candidate << StringRef(B, Len);
+        Candidate.clear();
+        InString = true;
+      } else if (P == E) {
+        Candidate.append(B, E);
       } else {
         Candidate.clear();
-        ++I;
-        ++Offset;
+      }
+      if (P == E) {
+        ChunkOffset += ChunkSize;
+        continue;
+      }
+      if (InString) {
+        OS << '\n';
+        InString = false;
+      }
+    }
+
+    const char *S = nullptr;
+    for (; P != E; ++P) {
+      if (isStringChar(*P)) {
+        if (!S)
+          S = P;
+      } else if (S) {
+        if (static_cast<size_t>(P - S) >= Min) {
+          printHeader(ChunkOffset + (S - B));
+          OS << StringRef(S, P - S) << '\n';
+        }
+        S = nullptr;
+      }
+    }
+
+    if (S) {
+      size_t Len = E - S;
+      if (Len >= Min) {
+        printHeader(ChunkOffset + (S - B));
+        OS << StringRef(S, Len);
+        InString = true;
+      } else {
+        Candidate.append(S, E);
       }
     }
+    ChunkOffset += ChunkSize;
   }
 
   if (InString)
@@ -224,7 +257,9 @@ int main(int argc, char **argv) {
       Expected<sys::fs::file_t> FDOrErr =
           sys::fs::openNativeFileForRead(File, sys::fs::OF_TextWithCRLF);
       if (!FDOrErr) {
-        errs() << File << ": " << toString(FDOrErr.takeError()) << '\n';
+        errs() << File
+               << ": cannot open file: " << toString(FDOrErr.takeError())
+               << '\n';
         continue;
       }
       strings(llvm::outs(), File, *FDOrErr);

>From d5c15cdd9c1e2a72da07879e7e1d68d3b9792aef Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Wed, 26 Aug 2026 20:59:07 -0500
Subject: [PATCH 09/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 .../tools/llvm-strings/chunk-boundary.test    |  2 +-
 llvm/tools/llvm-strings/llvm-strings.cpp      | 63 ++++++++++++-------
 2 files changed, 41 insertions(+), 24 deletions(-)

diff --git a/llvm/test/tools/llvm-strings/chunk-boundary.test b/llvm/test/tools/llvm-strings/chunk-boundary.test
index 9949bd76e51161..8ba6ecec979044 100644
--- a/llvm/test/tools/llvm-strings/chunk-boundary.test
+++ b/llvm/test/tools/llvm-strings/chunk-boundary.test
@@ -46,7 +46,7 @@ CASE5:{{^}}   4000 FRESH{{$}}
 RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16383 + b'S' + b'M' * 16384 + b'E' + b'\0' * 4)" > %t.6
 RUN: %python -c "import sys; sys.stdout.write('S' + 'M' * 16384 + 'E\n')" > %t.6.expected
 RUN: llvm-strings %t.6 > %t.6.out
-RUN: diff %t.6.expected %t.6.out
+RUN: diff --strip-trailing-cr %t.6.expected %t.6.out
 
 ## A string terminated by the end of the file (no trailing unprintable byte)
 ## must still be printed.
diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index c810ee090373ff..a228e612ddde5f 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -95,7 +95,7 @@ static bool isStringChar(char C) { return isPrint(C) || C == '\t'; }
 static void strings(raw_ostream &OS, StringRef FileName,
                     sys::fs::file_t Handle) {
   SmallString<sys::fs::DefaultReadChunkSize> Buffer;
-  auto printHeader = [&OS, FileName](unsigned StringStart) {
+  auto printHeader = [&OS, FileName](size_t StringStart) {
     if (PrintFileName)
       OS << FileName << ": ";
     switch (Radix) {
@@ -126,9 +126,13 @@ static void strings(raw_ostream &OS, StringRef FileName,
   SmallString<DefaultMinLength> Candidate;
   bool InString = false;
   // Offset of the start of the current chunk within the file.
-  unsigned ChunkOffset = 0;
+  size_t ChunkOffset = 0;
 
   Buffer.resize_for_overwrite(sys::fs::DefaultReadChunkSize);
+
+  // To prevent performance regression under O0, access the raw pointer instead
+  // of using methods provided by the standard library, which are not inlined
+  // under O0.
   while (true) {
     Expected<size_t> ReadBytesOrErr = sys::fs::readNativeFile(
         Handle, MutableArrayRef(Buffer.data(), Buffer.size()));
@@ -137,31 +141,40 @@ static void strings(raw_ostream &OS, StringRef FileName,
              << errorToErrorCode(ReadBytesOrErr.takeError()).message() << '\n';
       return;
     }
-    std::size_t ChunkSize = *ReadBytesOrErr;
+    size_t ChunkSize = *ReadBytesOrErr;
     if (ChunkSize == 0)
       break;
 
-    const char *const B = Buffer.data();
-    const char *const E = B + ChunkSize;
-    const char *P = B;
+    const char *const Begin = Buffer.data();
+    const char *const End = Begin + ChunkSize;
+    const char *Cur = Begin;
 
+    // Handle the remaining part from the previous chunk.
+    // The previous chunk can be either no longer than MinSize or part of the
+    // string.
+    while (Cur != End && isStringChar(*Cur))
+      ++Cur;
+    size_t Len = Cur - Begin;
+    // Keep the buffer size bounded. With a small Min, a long string spanning
+    // multiple chunks will have at most DefaultReadChunkSize bytes, since the
+    // buffer is printed immediately with the header (guarded by the second if).
+    // With a large Min, the buffer must hold at least Min bytes, since we need
+    // enough data to decide whether to print it.
     if (InString || !Candidate.empty()) {
-      while (P != E && isStringChar(*P))
-        ++P;
-      size_t Len = P - B;
       if (InString) {
-        OS << StringRef(B, Len);
+        OS << StringRef(Begin, Len);
       } else if (Candidate.size() + Len >= Min) {
         printHeader(ChunkOffset - Candidate.size());
-        OS << Candidate << StringRef(B, Len);
+        OS << Candidate << StringRef(Begin, Len);
         Candidate.clear();
         InString = true;
-      } else if (P == E) {
-        Candidate.append(B, E);
+      } else if (Cur == End) {
+        Candidate.append(Begin, End);
       } else {
         Candidate.clear();
       }
-      if (P == E) {
+
+      if (Cur == End) {
         ChunkOffset += ChunkSize;
         continue;
       }
@@ -172,27 +185,31 @@ static void strings(raw_ostream &OS, StringRef FileName,
     }
 
     const char *S = nullptr;
-    for (; P != E; ++P) {
-      if (isStringChar(*P)) {
+    for (; Cur != End; ++Cur) {
+      if (isStringChar(*Cur)) {
         if (!S)
-          S = P;
+          S = Cur;
       } else if (S) {
-        if (static_cast<size_t>(P - S) >= Min) {
-          printHeader(ChunkOffset + (S - B));
-          OS << StringRef(S, P - S) << '\n';
+        if (static_cast<size_t>(Cur - S) >= Min) {
+          printHeader(ChunkOffset + (S - Begin));
+          OS << StringRef(S, Cur - S) << '\n';
         }
         S = nullptr;
       }
     }
 
+    // Concatenate the last part.
+    // If the current buffer is no longer than Min, we can't print it, so just
+    // add it to the candidate buffer. If it is used, mark it as InString to
+    // prevent printing the header twice.
     if (S) {
-      size_t Len = E - S;
+      size_t Len = End - S;
       if (Len >= Min) {
-        printHeader(ChunkOffset + (S - B));
+        printHeader(ChunkOffset + (S - Begin));
         OS << StringRef(S, Len);
         InString = true;
       } else {
-        Candidate.append(S, E);
+        Candidate.append(S, End);
       }
     }
     ChunkOffset += ChunkSize;

>From fbdd37cf42be148592dfcb9f6cd3189fd96c2a55 Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Wed, 26 Aug 2026 23:11:36 -0500
Subject: [PATCH 10/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 llvm/tools/llvm-strings/llvm-strings.cpp | 6 +++---
 1 file changed, 3 insertions(+), 3 deletions(-)

diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index a228e612ddde5f..8ca24ae3c28431 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -152,15 +152,15 @@ static void strings(raw_ostream &OS, StringRef FileName,
     // Handle the remaining part from the previous chunk.
     // The previous chunk can be either no longer than MinSize or part of the
     // string.
-    while (Cur != End && isStringChar(*Cur))
-      ++Cur;
-    size_t Len = Cur - Begin;
     // Keep the buffer size bounded. With a small Min, a long string spanning
     // multiple chunks will have at most DefaultReadChunkSize bytes, since the
     // buffer is printed immediately with the header (guarded by the second if).
     // With a large Min, the buffer must hold at least Min bytes, since we need
     // enough data to decide whether to print it.
     if (InString || !Candidate.empty()) {
+      while (Cur != End && isStringChar(*Cur))
+        ++Cur;
+      size_t Len = Cur - Begin;
       if (InString) {
         OS << StringRef(Begin, Len);
       } else if (Candidate.size() + Len >= Min) {

>From 3c0b3d32008c1cb8b6bf2b08b630ed98d8997189 Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Thu, 27 Aug 2026 11:33:39 -0500
Subject: [PATCH 11/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 .../tools/llvm-strings/chunk-boundary.test    |  4 +-
 llvm/tools/llvm-strings/llvm-strings.cpp      | 59 +++++++++++++------
 2 files changed, 43 insertions(+), 20 deletions(-)

diff --git a/llvm/test/tools/llvm-strings/chunk-boundary.test b/llvm/test/tools/llvm-strings/chunk-boundary.test
index 8ba6ecec979044..91ee9ec15eb270 100644
--- a/llvm/test/tools/llvm-strings/chunk-boundary.test
+++ b/llvm/test/tools/llvm-strings/chunk-boundary.test
@@ -44,9 +44,9 @@ CASE5:{{^}}   4000 FRESH{{$}}
 ## intact, once. The expected output is generated rather than written as a
 ## CHECK line because it is a chunk long.
 RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16383 + b'S' + b'M' * 16384 + b'E' + b'\0' * 4)" > %t.6
-RUN: %python -c "import sys; sys.stdout.write('S' + 'M' * 16384 + 'E\n')" > %t.6.expected
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'S' + b'M' * 16384 + b'E\n')" > %t.6.expected
 RUN: llvm-strings %t.6 > %t.6.out
-RUN: diff --strip-trailing-cr %t.6.expected %t.6.out
+RUN: diff %t.6.expected %t.6.out
 
 ## A string terminated by the end of the file (no trailing unprintable byte)
 ## must still be printed.
diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 8ca24ae3c28431..b2ac994a90d766 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -150,7 +150,7 @@ static void strings(raw_ostream &OS, StringRef FileName,
     const char *Cur = Begin;
 
     // Handle the remaining part from the previous chunk.
-    // The previous chunk can be either no longer than MinSize or part of the
+    // The previous chunk can be either shorter than MinSize or part of the
     // string.
     // Keep the buffer size bounded. With a small Min, a long string spanning
     // multiple chunks will have at most DefaultReadChunkSize bytes, since the
@@ -158,58 +158,81 @@ static void strings(raw_ostream &OS, StringRef FileName,
     // With a large Min, the buffer must hold at least Min bytes, since we need
     // enough data to decide whether to print it.
     if (InString || !Candidate.empty()) {
+      // Find the end of the current buffer
       while (Cur != End && isStringChar(*Cur))
         ++Cur;
       size_t Len = Cur - Begin;
       if (InString) {
+        // Print the remaining part if the previous chunk has already printed
+        // the header. E.g. header: aaaaa | bbbbb, where | is the chunk
+        // boundary.                        ^
         OS << StringRef(Begin, Len);
       } else if (Candidate.size() + Len >= Min) {
+        // If the header hasn't been printed yet (e.g. the previous candidate
+        // was smaller than Min), but we can print it now, print the header
+        // first, followed by the candidate from the previous chunk and the
+        // current string. E.g. '\0' | bbbbbb
+        //                             ^
         printHeader(ChunkOffset - Candidate.size());
         OS << Candidate << StringRef(Begin, Len);
         Candidate.clear();
         InString = true;
       } else if (Cur == End) {
+        // If the current chunk + previous candidate is still smaller than Min ,
+        // append it to Candidate
         Candidate.append(Begin, End);
       } else {
+        // If the string has terminated but is still smaller than Min, clear the
+        // buffer since it is too short to print.
         Candidate.clear();
       }
 
       if (Cur == End) {
+        // Finish handling the current chunk and update ChunkOffset.
         ChunkOffset += ChunkSize;
         continue;
       }
       if (InString) {
+        // We haven't reached the end of the chunk, which means the string is
+        // terminated. Add a '\n' to start printing a new string.
         OS << '\n';
         InString = false;
       }
     }
 
-    const char *S = nullptr;
+    // At this point, we are always at the start of a new string because the
+    // remaining part of the previous string has already been handled.
+    const char *StrHead = nullptr;
     for (; Cur != End; ++Cur) {
       if (isStringChar(*Cur)) {
-        if (!S)
-          S = Cur;
-      } else if (S) {
-        if (static_cast<size_t>(Cur - S) >= Min) {
-          printHeader(ChunkOffset + (S - Begin));
-          OS << StringRef(S, Cur - S) << '\n';
+        // Find the start of the next string
+        if (!StrHead)
+          StrHead = Cur;
+      } else if (StrHead) {
+        // If it is not a StringChar, we have reached the end of the current
+        // string. Try to print it.
+        if (static_cast<size_t>(Cur - StrHead) >= Min) {
+          printHeader(ChunkOffset + (StrHead - Begin));
+          OS << StringRef(StrHead, Cur - StrHead) << '\n';
         }
-        S = nullptr;
+        StrHead = nullptr;
       }
     }
 
-    // Concatenate the last part.
-    // If the current buffer is no longer than Min, we can't print it, so just
-    // add it to the candidate buffer. If it is used, mark it as InString to
-    // prevent printing the header twice.
-    if (S) {
-      size_t Len = End - S;
+    // The last string spans multiple chunks. If it is larger than Min, print
+    // the header immediately and set the InString flag to avoid printing it
+    // again.
+    // e.g. aaaaa | bbbbb
+    //          ^
+    if (StrHead) {
+      size_t Len = End - StrHead;
+      // Print it, or append it to Candidate if it is too short.
       if (Len >= Min) {
-        printHeader(ChunkOffset + (S - Begin));
-        OS << StringRef(S, Len);
+        printHeader(ChunkOffset + (StrHead - Begin));
+        OS << StringRef(StrHead, Len);
         InString = true;
       } else {
-        Candidate.append(S, End);
+        Candidate.append(StrHead, End);
       }
     }
     ChunkOffset += ChunkSize;

>From ffa54c058465416e5e9b5e355d96d2b51f0732ed Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Sun, 13 Sep 2026 15:34:20 -0500
Subject: [PATCH 12/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 .../tools/llvm-strings/chunk-boundary.test    | 18 ++++++++--
 llvm/tools/llvm-strings/llvm-strings.cpp      | 33 +++++++++----------
 2 files changed, 32 insertions(+), 19 deletions(-)

diff --git a/llvm/test/tools/llvm-strings/chunk-boundary.test b/llvm/test/tools/llvm-strings/chunk-boundary.test
index 91ee9ec15eb270..7a644fab0c4e60 100644
--- a/llvm/test/tools/llvm-strings/chunk-boundary.test
+++ b/llvm/test/tools/llvm-strings/chunk-boundary.test
@@ -48,9 +48,23 @@ RUN: %python -c "import sys; sys.stdout.buffer.write(b'S' + b'M' * 16384 + b'E\n
 RUN: llvm-strings %t.6 > %t.6.out
 RUN: diff %t.6.expected %t.6.out
 
+## Case 7: the minimum length is greater than the chunk size, so a candidate
+## has to be buffered across more than one chunk before it is known to be long
+## enough. A run of exactly the minimum length is printed, with the offset of
+## its start (0x8), which precedes the chunk in which the decision is made.
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 8 + b'A' * 20000 + b'\0' * 4)" > %t.7
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'%7x ' % 8 + b'A' * 20000 + b'\n')" > %t.7.expected
+RUN: llvm-strings --radix=x --bytes=20000 %t.7 > %t.7.out
+RUN: diff %t.7.expected %t.7.out
+
+## Case 8: as case 7, but the run is one byte short of the minimum length, so
+## the buffered candidate is discarded and nothing is printed.
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 8 + b'A' * 19999 + b'\0' * 4)" > %t.8
+RUN: llvm-strings --bytes=20000 %t.8 | count 0
+
 ## A string terminated by the end of the file (no trailing unprintable byte)
 ## must still be printed.
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16380 + b'TRAILING')" > %t.7
-RUN: llvm-strings %t.7 | FileCheck %s --check-prefix=EOF --strict-whitespace --implicit-check-not={{.}}
+RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16380 + b'TRAILING')" > %t.9
+RUN: llvm-strings %t.9 | FileCheck %s --check-prefix=EOF --strict-whitespace --implicit-check-not={{.}}
 
 EOF:{{^}}TRAILING{{$}}
diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index b2ac994a90d766..6f170cd008eff3 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -130,9 +130,6 @@ static void strings(raw_ostream &OS, StringRef FileName,
 
   Buffer.resize_for_overwrite(sys::fs::DefaultReadChunkSize);
 
-  // To prevent performance regression under O0, access the raw pointer instead
-  // of using methods provided by the standard library, which are not inlined
-  // under O0.
   while (true) {
     Expected<size_t> ReadBytesOrErr = sys::fs::readNativeFile(
         Handle, MutableArrayRef(Buffer.data(), Buffer.size()));
@@ -145,6 +142,9 @@ static void strings(raw_ostream &OS, StringRef FileName,
     if (ChunkSize == 0)
       break;
 
+    // To prevent performance regression under O0, access the raw pointer
+    // instead of using methods provided by the standard library, which are not
+    // inlined under O0.
     const char *const Begin = Buffer.data();
     const char *const End = Begin + ChunkSize;
     const char *Cur = Begin;
@@ -158,28 +158,29 @@ static void strings(raw_ostream &OS, StringRef FileName,
     // With a large Min, the buffer must hold at least Min bytes, since we need
     // enough data to decide whether to print it.
     if (InString || !Candidate.empty()) {
-      // Find the end of the current buffer
+      // Find the end of the current string.
       while (Cur != End && isStringChar(*Cur))
         ++Cur;
       size_t Len = Cur - Begin;
       if (InString) {
         // Print the remaining part if the previous chunk has already printed
         // the header. E.g. header: aaaaa | bbbbb, where | is the chunk
-        // boundary.                        ^
+        // boundary.
+        // Output: Header: aaaaabbbbb, where bbbbb is printed in here.
         OS << StringRef(Begin, Len);
       } else if (Candidate.size() + Len >= Min) {
         // If the header hasn't been printed yet (e.g. the previous candidate
         // was smaller than Min), but we can print it now, print the header
         // first, followed by the candidate from the previous chunk and the
-        // current string. E.g. '\0' | bbbbbb
-        //                             ^
+        // current string. E.g. aa | bbbbbb
+        // Output Header: aabbbbbb, where aabbbbbb is printed in here.
         printHeader(ChunkOffset - Candidate.size());
         OS << Candidate << StringRef(Begin, Len);
         Candidate.clear();
         InString = true;
       } else if (Cur == End) {
-        // If the current chunk + previous candidate is still smaller than Min ,
-        // append it to Candidate
+        // If the current chunk + previous candidate is still smaller than Min,
+        // append it to Candidate.
         Candidate.append(Begin, End);
       } else {
         // If the string has terminated but is still smaller than Min, clear the
@@ -205,12 +206,12 @@ static void strings(raw_ostream &OS, StringRef FileName,
     const char *StrHead = nullptr;
     for (; Cur != End; ++Cur) {
       if (isStringChar(*Cur)) {
-        // Find the start of the next string
+        // Find the start of the next string.
         if (!StrHead)
           StrHead = Cur;
       } else if (StrHead) {
-        // If it is not a StringChar, we have reached the end of the current
-        // string. Try to print it.
+        // If it is not a printable character, we have reached the end of the
+        // current string. Print it if long enough.
         if (static_cast<size_t>(Cur - StrHead) >= Min) {
           printHeader(ChunkOffset + (StrHead - Begin));
           OS << StringRef(StrHead, Cur - StrHead) << '\n';
@@ -219,11 +220,9 @@ static void strings(raw_ostream &OS, StringRef FileName,
       }
     }
 
-    // The last string spans multiple chunks. If it is larger than Min, print
-    // the header immediately and set the InString flag to avoid printing it
-    // again.
-    // e.g. aaaaa | bbbbb
-    //          ^
+    // The last string could span multiple chunks. If it is larger than Min,
+    // print the header immediately and set the InString flag to avoid printing
+    // it again.
     if (StrHead) {
       size_t Len = End - StrHead;
       // Print it, or append it to Candidate if it is too short.

>From 8296a37f115d23e45e115d377d9ea1405f3520ae Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Sun, 13 Sep 2026 16:25:20 -0500
Subject: [PATCH 13/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 llvm/tools/llvm-strings/llvm-strings.cpp | 19 ++++---------------
 1 file changed, 4 insertions(+), 15 deletions(-)

diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 4109842bb07833..97dd8134a21bbc 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -95,7 +95,7 @@ static bool isStringChar(char C) { return isPrint(C) || C == '\t'; }
 static void strings(raw_ostream &OS, StringRef FileName,
                     sys::fs::file_t Handle) {
   SmallString<sys::fs::DefaultReadChunkSize> Buffer;
-  auto printHeader = [&OS, FileName](size_t StringStart) {
+  auto PrintHeader = [&OS, FileName](size_t StringStart) {
     if (PrintFileName)
       OS << FileName << ": ";
     switch (Radix) {
@@ -137,17 +137,6 @@ static void strings(raw_ostream &OS, StringRef FileName,
       errs() << FileName << ": "
              << errorToErrorCode(ReadBytesOrErr.takeError()).message() << '\n';
       return;
-=======
-  const char *B = Contents.begin();
-  const char *P = nullptr, *E = nullptr, *S = nullptr;
-  for (P = Contents.begin(), E = Contents.end(); P < E; ++P) {
-    if (isPrint(*P) || *P == '\t') {
-      if (S == nullptr)
-        S = P;
-    } else if (S) {
-      Print(S - B, StringRef(S, P - S));
-      S = nullptr;
->>>>>>> main
     }
     size_t ChunkSize = *ReadBytesOrErr;
     if (ChunkSize == 0)
@@ -185,7 +174,7 @@ static void strings(raw_ostream &OS, StringRef FileName,
         // first, followed by the candidate from the previous chunk and the
         // current string. E.g. aa | bbbbbb
         // Output Header: aabbbbbb, where aabbbbbb is printed in here.
-        printHeader(ChunkOffset - Candidate.size());
+        PrintHeader(ChunkOffset - Candidate.size());
         OS << Candidate << StringRef(Begin, Len);
         Candidate.clear();
         InString = true;
@@ -224,7 +213,7 @@ static void strings(raw_ostream &OS, StringRef FileName,
         // If it is not a printable character, we have reached the end of the
         // current string. Print it if long enough.
         if (static_cast<size_t>(Cur - StrHead) >= Min) {
-          printHeader(ChunkOffset + (StrHead - Begin));
+          PrintHeader(ChunkOffset + (StrHead - Begin));
           OS << StringRef(StrHead, Cur - StrHead) << '\n';
         }
         StrHead = nullptr;
@@ -238,7 +227,7 @@ static void strings(raw_ostream &OS, StringRef FileName,
       size_t Len = End - StrHead;
       // Print it, or append it to Candidate if it is too short.
       if (Len >= Min) {
-        printHeader(ChunkOffset + (StrHead - Begin));
+        PrintHeader(ChunkOffset + (StrHead - Begin));
         OS << StringRef(StrHead, Len);
         InString = true;
       } else {

>From 41ae0922cfb4eb71637989e573081482fa1b6947 Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Tue, 15 Sep 2026 15:54:06 -0500
Subject: [PATCH 14/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 .../tools/llvm-strings/chunk-boundary.test    | 103 ++++++++++++------
 1 file changed, 70 insertions(+), 33 deletions(-)

diff --git a/llvm/test/tools/llvm-strings/chunk-boundary.test b/llvm/test/tools/llvm-strings/chunk-boundary.test
index 7a644fab0c4e60..0d8652ec1e9fb6 100644
--- a/llvm/test/tools/llvm-strings/chunk-boundary.test
+++ b/llvm/test/tools/llvm-strings/chunk-boundary.test
@@ -1,70 +1,107 @@
 ## Show that strings interacting with the read-chunk boundary are reported
 ## correctly. The input files are crafted assuming the native read chunk size
 ## (sys::fs::DefaultReadChunkSize) of 16384 bytes.
+##
+## The inputs, and the expected output lines that are too long to write as
+## CHECK lines, are generated by the Python script at the end of this file,
+## which is run as "%python %s". Each argument describes one part of the
+## output: a bare number is that many zero bytes, TEXT at N is TEXT repeated N
+## times, and anything else is literal text. With --line the parts are written
+## as a single line of llvm-strings output instead, preceded by the offset
+## header if --offset is given.
 
 ## Case 1: at least min string size appears before the boundary, unprintable
 ## byte as first byte of the next chunk. The string is printed on its own,
-## with the offset of its start (0x3ff8 = 16384 - 8).
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16376 + b'ENDCHUNK' + b'\0' * 4)" > %t.1
-RUN: llvm-strings --radix=x %t.1 | FileCheck %s --check-prefix=CASE1 --strict-whitespace --implicit-check-not={{.}}
+## with the offset of its start (16376 = 16384 - 8).
+# RUN: %python %s 16376 ENDCHUNK 4 > %t.1
+# RUN: llvm-strings --radix=d %t.1 | FileCheck %s --check-prefix=CASE1 --match-full-lines --strict-whitespace --implicit-check-not={{.}}
 
-CASE1:{{^}}   3ff8 ENDCHUNK{{$}}
+# CASE1:  16376 ENDCHUNK
 
 ## Case 2: at least min string size appears before the boundary, printable
 ## byte as first byte of the next chunk. The prefix is printed together with
 ## the following characters, as a single string.
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16378 + b'BEFORE' + b'AFTER!' + b'\0' * 4)" > %t.2
-RUN: llvm-strings --radix=x %t.2 | FileCheck %s --check-prefix=CASE2 --strict-whitespace --implicit-check-not={{.}}
+# RUN: %python %s 16378 BEFORE AFTER! 4 > %t.2
+# RUN: llvm-strings --radix=d %t.2 | FileCheck %s --check-prefix=CASE2 --match-full-lines --strict-whitespace --implicit-check-not={{.}}
 
-CASE2:{{^}}   3ffa BEFOREAFTER!{{$}}
+# CASE2:  16378 BEFOREAFTER!
 
 ## Case 3: less than min string size appears before the boundary, unprintable
 ## byte as the next byte. The prefix is not printed.
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16382 + b'AB' + b'\0' * 4)" > %t.3
-RUN: llvm-strings %t.3 | count 0
+# RUN: %python %s 16382 AB 4 > %t.3
+# RUN: llvm-strings %t.3 | count 0
 
 ## Case 4: less than min string size appears before the boundary, printable
 ## bytes as the next bytes, forming a min length string. The prefix is
 ## printed together with the following characters, with the offset of its
-## true start (0x3ffd = 16384 - 3).
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16381 + b'ABC' + b'DEF' + b'\0' * 4)" > %t.4
-RUN: llvm-strings --radix=x %t.4 | FileCheck %s --check-prefix=CASE4 --strict-whitespace --implicit-check-not={{.}}
+## true start (16381 = 16384 - 3).
+# RUN: %python %s 16381 ABC DEF 4 > %t.4
+# RUN: llvm-strings --radix=d %t.4 | FileCheck %s --check-prefix=CASE4 --match-full-lines --strict-whitespace --implicit-check-not={{.}}
 
-CASE4:{{^}}   3ffd ABCDEF{{$}}
+# CASE4:  16381 ABCDEF
 
 ## Case 5: the prefix is empty at the start of a chunk that starts with a min
-## length string (0x4000 = 16384).
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16384 + b'FRESH' + b'\0' * 4)" > %t.5
-RUN: llvm-strings --radix=x %t.5 | FileCheck %s --check-prefix=CASE5 --strict-whitespace --implicit-check-not={{.}}
+## length string (16384).
+# RUN: %python %s 16384 FRESH 4 > %t.5
+# RUN: llvm-strings --radix=d %t.5 | FileCheck %s --check-prefix=CASE5 --match-full-lines --strict-whitespace --implicit-check-not={{.}}
 
-CASE5:{{^}}   4000 FRESH{{$}}
+# CASE5:  16384 FRESH
 
 ## Case 6: a string that spans the entirety of one chunk, with one character
 ## before the chunk start and one after its end. The string is printed
 ## intact, once. The expected output is generated rather than written as a
 ## CHECK line because it is a chunk long.
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16383 + b'S' + b'M' * 16384 + b'E' + b'\0' * 4)" > %t.6
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'S' + b'M' * 16384 + b'E\n')" > %t.6.expected
-RUN: llvm-strings %t.6 > %t.6.out
-RUN: diff %t.6.expected %t.6.out
+# RUN: %python %s 16383 S M at 16384 E 4 > %t.6
+# RUN: %python %s --line S M at 16384 E > %t.6.expected
+# RUN: llvm-strings %t.6 > %t.6.out
+# RUN: diff %t.6.expected %t.6.out
 
 ## Case 7: the minimum length is greater than the chunk size, so a candidate
 ## has to be buffered across more than one chunk before it is known to be long
 ## enough. A run of exactly the minimum length is printed, with the offset of
-## its start (0x8), which precedes the chunk in which the decision is made.
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 8 + b'A' * 20000 + b'\0' * 4)" > %t.7
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'%7x ' % 8 + b'A' * 20000 + b'\n')" > %t.7.expected
-RUN: llvm-strings --radix=x --bytes=20000 %t.7 > %t.7.out
-RUN: diff %t.7.expected %t.7.out
+## its start, which lies in an earlier chunk than the one where it is printed.
+# RUN: %python %s 8 A at 20000 4 > %t.7
+# RUN: %python %s --line --offset=8 A at 20000 > %t.7.expected
+# RUN: llvm-strings --radix=d --bytes=20000 %t.7 > %t.7.out
+# RUN: diff %t.7.expected %t.7.out
 
 ## Case 8: as case 7, but the run is one byte short of the minimum length, so
 ## the buffered candidate is discarded and nothing is printed.
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 8 + b'A' * 19999 + b'\0' * 4)" > %t.8
-RUN: llvm-strings --bytes=20000 %t.8 | count 0
+# RUN: %python %s 8 A at 19999 4 > %t.8
+# RUN: llvm-strings --bytes=20000 %t.8 | count 0
 
-## A string terminated by the end of the file (no trailing unprintable byte)
-## must still be printed.
-RUN: %python -c "import sys; sys.stdout.buffer.write(b'\0' * 16380 + b'TRAILING')" > %t.9
-RUN: llvm-strings %t.9 | FileCheck %s --check-prefix=EOF --strict-whitespace --implicit-check-not={{.}}
+## Case 9: a string is terminated by the end of the file, with no trailing
+## unprintable byte. The string must still be printed.
+# RUN: %python %s 16380 TRAILING > %t.9
+# RUN: llvm-strings %t.9 | FileCheck %s --check-prefix=CASE9 --match-full-lines --strict-whitespace --implicit-check-not={{.}}
 
-EOF:{{^}}TRAILING{{$}}
+# CASE9:TRAILING
+
+import sys
+
+Args = sys.argv[1:]
+Line = False
+Offset = None
+while Args and Args[0].startswith("--"):
+    Opt = Args.pop(0)
+    if Opt == "--line":
+        Line = True
+    elif Opt.startswith("--offset="):
+        Offset = int(Opt[len("--offset=") :])
+    else:
+        sys.exit("unknown option: " + Opt)
+
+Parts = []
+for Arg in Args:
+    if Arg.isdigit():
+        Parts.append(b"\0" * int(Arg))
+    else:
+        Text, At, Count = Arg.partition("@")
+        Parts.append(Text.encode() * (int(Count) if At else 1))
+
+Out = b"".join(Parts)
+if Line:
+    if Offset is not None:
+        Out = b"%7d " % Offset + Out
+    Out += b"\n"
+sys.stdout.buffer.write(Out)

>From ed83a4459edfbf2890f961e4c37c672ec187e032 Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Wed, 16 Sep 2026 16:34:55 -0500
Subject: [PATCH 15/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 .../tools/llvm-strings/chunk-boundary.test    | 87 +++++++++++--------
 1 file changed, 49 insertions(+), 38 deletions(-)

diff --git a/llvm/test/tools/llvm-strings/chunk-boundary.test b/llvm/test/tools/llvm-strings/chunk-boundary.test
index 0d8652ec1e9fb6..89d886926f3f87 100644
--- a/llvm/test/tools/llvm-strings/chunk-boundary.test
+++ b/llvm/test/tools/llvm-strings/chunk-boundary.test
@@ -2,13 +2,8 @@
 ## correctly. The input files are crafted assuming the native read chunk size
 ## (sys::fs::DefaultReadChunkSize) of 16384 bytes.
 ##
-## The inputs, and the expected output lines that are too long to write as
-## CHECK lines, are generated by the Python script at the end of this file,
-## which is run as "%python %s". Each argument describes one part of the
-## output: a bare number is that many zero bytes, TEXT at N is TEXT repeated N
-## times, and anything else is literal text. With --line the parts are written
-## as a single line of llvm-strings output instead, preceded by the offset
-## header if --offset is given.
+## The inputs, and expected output lines too long to write as CHECK lines, are
+## generated by the script at the end of this file.
 
 ## Case 1: at least min string size appears before the boundary, unprintable
 ## byte as first byte of the next chunk. The string is printed on its own,
@@ -51,8 +46,8 @@
 ## before the chunk start and one after its end. The string is printed
 ## intact, once. The expected output is generated rather than written as a
 ## CHECK line because it is a chunk long.
-# RUN: %python %s 16383 S M at 16384 E 4 > %t.6
-# RUN: %python %s --line S M at 16384 E > %t.6.expected
+# RUN: %python %s 16383 S M E 4 --repeat 1,1,16384,1,1 > %t.6
+# RUN: %python %s --line S M E --repeat 1,16384,1 > %t.6.expected
 # RUN: llvm-strings %t.6 > %t.6.out
 # RUN: diff %t.6.expected %t.6.out
 
@@ -60,14 +55,14 @@
 ## has to be buffered across more than one chunk before it is known to be long
 ## enough. A run of exactly the minimum length is printed, with the offset of
 ## its start, which lies in an earlier chunk than the one where it is printed.
-# RUN: %python %s 8 A at 20000 4 > %t.7
-# RUN: %python %s --line --offset=8 A at 20000 > %t.7.expected
+# RUN: %python %s 8 A 4 --repeat 1,20000,1 > %t.7
+# RUN: %python %s --line --offset=8 A --repeat 20000 > %t.7.expected
 # RUN: llvm-strings --radix=d --bytes=20000 %t.7 > %t.7.out
 # RUN: diff %t.7.expected %t.7.out
 
 ## Case 8: as case 7, but the run is one byte short of the minimum length, so
 ## the buffered candidate is discarded and nothing is printed.
-# RUN: %python %s 8 A at 19999 4 > %t.8
+# RUN: %python %s 8 A 4 --repeat 1,19999,1 > %t.8
 # RUN: llvm-strings --bytes=20000 %t.8 | count 0
 
 ## Case 9: a string is terminated by the end of the file, with no trailing
@@ -77,31 +72,47 @@
 
 # CASE9:TRAILING
 
+import argparse
 import sys
 
-Args = sys.argv[1:]
-Line = False
-Offset = None
-while Args and Args[0].startswith("--"):
-    Opt = Args.pop(0)
-    if Opt == "--line":
-        Line = True
-    elif Opt.startswith("--offset="):
-        Offset = int(Opt[len("--offset=") :])
-    else:
-        sys.exit("unknown option: " + Opt)
-
-Parts = []
-for Arg in Args:
-    if Arg.isdigit():
-        Parts.append(b"\0" * int(Arg))
-    else:
-        Text, At, Count = Arg.partition("@")
-        Parts.append(Text.encode() * (int(Count) if At else 1))
-
-Out = b"".join(Parts)
-if Line:
-    if Offset is not None:
-        Out = b"%7d " % Offset + Out
-    Out += b"\n"
-sys.stdout.buffer.write(Out)
+parser = argparse.ArgumentParser(description="Write the concatenation of PARTs.")
+parser.add_argument(
+    "parts",
+    nargs="*",
+    metavar="PART",
+    help="a number, for that many zero bytes, or else literal text",
+)
+parser.add_argument(
+    "--repeat",
+    type=lambda arg: [int(n) for n in arg.split(",")] if arg else [],
+    default=[],
+    help="comma-separated number of times to write each PART (default: once)",
+)
+parser.add_argument(
+    "--line",
+    action="store_true",
+    help="write the PARTs as a single line of llvm-strings output",
+)
+parser.add_argument(
+    "--offset",
+    type=int,
+    help="with --line, precede the line with OFFSET as printed by --radix=d",
+)
+args = parser.parse_intermixed_args()
+
+repeat = args.repeat or [1] * len(args.parts)
+if len(repeat) != len(args.parts):
+    parser.error(
+        "--repeat has %d counts for %d PARTs" % (len(repeat), len(args.parts))
+    )
+
+out = b"".join(
+    (b"\0" * int(part) if part.isdigit() else part.encode()) * count
+    for part, count in zip(args.parts, repeat)
+)
+if args.line:
+    if args.offset is not None:
+        out = b"%7d " % args.offset + out
+    out += b"\n"
+# Write bytes, not text, so that "\n" is not written as "\r\n" on Windows.
+sys.stdout.buffer.write(out)

>From 240b2fc31961429cca279c04e803f80de75acbe6 Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Tue, 22 Sep 2026 12:50:19 -0500
Subject: [PATCH 16/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 .../tools/llvm-strings/{chunk-boundary.test => chunk-boundary.py} | 0
 1 file changed, 0 insertions(+), 0 deletions(-)
 rename llvm/test/tools/llvm-strings/{chunk-boundary.test => chunk-boundary.py} (100%)

diff --git a/llvm/test/tools/llvm-strings/chunk-boundary.test b/llvm/test/tools/llvm-strings/chunk-boundary.py
similarity index 100%
rename from llvm/test/tools/llvm-strings/chunk-boundary.test
rename to llvm/test/tools/llvm-strings/chunk-boundary.py

>From d87dff3cadd6442cef28a29799cae79a6c71f60e Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Tue, 22 Sep 2026 13:00:31 -0500
Subject: [PATCH 17/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 llvm/test/tools/llvm-strings/chunk-boundary.py | 4 +---
 1 file changed, 1 insertion(+), 3 deletions(-)

diff --git a/llvm/test/tools/llvm-strings/chunk-boundary.py b/llvm/test/tools/llvm-strings/chunk-boundary.py
index 89d886926f3f87..9d0641675c87b9 100644
--- a/llvm/test/tools/llvm-strings/chunk-boundary.py
+++ b/llvm/test/tools/llvm-strings/chunk-boundary.py
@@ -102,9 +102,7 @@
 
 repeat = args.repeat or [1] * len(args.parts)
 if len(repeat) != len(args.parts):
-    parser.error(
-        "--repeat has %d counts for %d PARTs" % (len(repeat), len(args.parts))
-    )
+    parser.error("--repeat has %d counts for %d PARTs" % (len(repeat), len(args.parts)))
 
 out = b"".join(
     (b"\0" * int(part) if part.isdigit() else part.encode()) * count

>From 04ae805ab847a4dc5807d0d6158ac466c003c52a Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Wed, 23 Sep 2026 12:10:29 -0500
Subject: [PATCH 18/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 ...{chunk-boundary.py => chunk-boundary.test} | 72 ++++---------------
 1 file changed, 13 insertions(+), 59 deletions(-)
 rename llvm/test/tools/llvm-strings/{chunk-boundary.py => chunk-boundary.test} (59%)

diff --git a/llvm/test/tools/llvm-strings/chunk-boundary.py b/llvm/test/tools/llvm-strings/chunk-boundary.test
similarity index 59%
rename from llvm/test/tools/llvm-strings/chunk-boundary.py
rename to llvm/test/tools/llvm-strings/chunk-boundary.test
index 9d0641675c87b9..658a7b1cabbb9d 100644
--- a/llvm/test/tools/llvm-strings/chunk-boundary.py
+++ b/llvm/test/tools/llvm-strings/chunk-boundary.test
@@ -1,14 +1,11 @@
 ## Show that strings interacting with the read-chunk boundary are reported
 ## correctly. The input files are crafted assuming the native read chunk size
 ## (sys::fs::DefaultReadChunkSize) of 16384 bytes.
-##
-## The inputs, and expected output lines too long to write as CHECK lines, are
-## generated by the script at the end of this file.
 
 ## Case 1: at least min string size appears before the boundary, unprintable
 ## byte as first byte of the next chunk. The string is printed on its own,
 ## with the offset of its start (16376 = 16384 - 8).
-# RUN: %python %s 16376 ENDCHUNK 4 > %t.1
+# RUN: printf "%16376sENDCHUNK%4s" | tr " " "\0" > %t.1
 # RUN: llvm-strings --radix=d %t.1 | FileCheck %s --check-prefix=CASE1 --match-full-lines --strict-whitespace --implicit-check-not={{.}}
 
 # CASE1:  16376 ENDCHUNK
@@ -16,28 +13,28 @@
 ## Case 2: at least min string size appears before the boundary, printable
 ## byte as first byte of the next chunk. The prefix is printed together with
 ## the following characters, as a single string.
-# RUN: %python %s 16378 BEFORE AFTER! 4 > %t.2
+# RUN: printf "%16378sBEFOREAFTER!%4s" | tr " " "\0" > %t.2
 # RUN: llvm-strings --radix=d %t.2 | FileCheck %s --check-prefix=CASE2 --match-full-lines --strict-whitespace --implicit-check-not={{.}}
 
 # CASE2:  16378 BEFOREAFTER!
 
 ## Case 3: less than min string size appears before the boundary, unprintable
 ## byte as the next byte. The prefix is not printed.
-# RUN: %python %s 16382 AB 4 > %t.3
+# RUN: printf "%16382sAB%4s" | tr " " "\0" > %t.3
 # RUN: llvm-strings %t.3 | count 0
 
 ## Case 4: less than min string size appears before the boundary, printable
 ## bytes as the next bytes, forming a min length string. The prefix is
 ## printed together with the following characters, with the offset of its
 ## true start (16381 = 16384 - 3).
-# RUN: %python %s 16381 ABC DEF 4 > %t.4
+# RUN: printf "%16381sABCDEF%4s" | tr " " "\0" > %t.4
 # RUN: llvm-strings --radix=d %t.4 | FileCheck %s --check-prefix=CASE4 --match-full-lines --strict-whitespace --implicit-check-not={{.}}
 
 # CASE4:  16381 ABCDEF
 
 ## Case 5: the prefix is empty at the start of a chunk that starts with a min
 ## length string (16384).
-# RUN: %python %s 16384 FRESH 4 > %t.5
+# RUN: printf "%16384sFRESH%4s" | tr " " "\0" > %t.5
 # RUN: llvm-strings --radix=d %t.5 | FileCheck %s --check-prefix=CASE5 --match-full-lines --strict-whitespace --implicit-check-not={{.}}
 
 # CASE5:  16384 FRESH
@@ -46,71 +43,28 @@
 ## before the chunk start and one after its end. The string is printed
 ## intact, once. The expected output is generated rather than written as a
 ## CHECK line because it is a chunk long.
-# RUN: %python %s 16383 S M E 4 --repeat 1,1,16384,1,1 > %t.6
-# RUN: %python %s --line S M E --repeat 1,16384,1 > %t.6.expected
+# RUN: printf "%16383sS%.16384dE%4s" | tr " " "\0" > %t.6
+# RUN: printf "S%.16384dE\n" > %t.6.expected
 # RUN: llvm-strings %t.6 > %t.6.out
-# RUN: diff %t.6.expected %t.6.out
+# RUN: diff --strip-trailing-cr %t.6.expected %t.6.out
 
 ## Case 7: the minimum length is greater than the chunk size, so a candidate
 ## has to be buffered across more than one chunk before it is known to be long
 ## enough. A run of exactly the minimum length is printed, with the offset of
 ## its start, which lies in an earlier chunk than the one where it is printed.
-# RUN: %python %s 8 A 4 --repeat 1,20000,1 > %t.7
-# RUN: %python %s --line --offset=8 A --repeat 20000 > %t.7.expected
+# RUN: printf "%8s%.20000d%4s" | tr " " "\0" > %t.7
+# RUN: printf "      8 %.20000d\n" > %t.7.expected
 # RUN: llvm-strings --radix=d --bytes=20000 %t.7 > %t.7.out
-# RUN: diff %t.7.expected %t.7.out
+# RUN: diff --strip-trailing-cr %t.7.expected %t.7.out
 
 ## Case 8: as case 7, but the run is one byte short of the minimum length, so
 ## the buffered candidate is discarded and nothing is printed.
-# RUN: %python %s 8 A 4 --repeat 1,19999,1 > %t.8
+# RUN: printf "%8s%.19999d%4s" | tr " " "\0" > %t.8
 # RUN: llvm-strings --bytes=20000 %t.8 | count 0
 
 ## Case 9: a string is terminated by the end of the file, with no trailing
 ## unprintable byte. The string must still be printed.
-# RUN: %python %s 16380 TRAILING > %t.9
+# RUN: printf "%16380sTRAILING" | tr " " "\0" > %t.9
 # RUN: llvm-strings %t.9 | FileCheck %s --check-prefix=CASE9 --match-full-lines --strict-whitespace --implicit-check-not={{.}}
 
 # CASE9:TRAILING
-
-import argparse
-import sys
-
-parser = argparse.ArgumentParser(description="Write the concatenation of PARTs.")
-parser.add_argument(
-    "parts",
-    nargs="*",
-    metavar="PART",
-    help="a number, for that many zero bytes, or else literal text",
-)
-parser.add_argument(
-    "--repeat",
-    type=lambda arg: [int(n) for n in arg.split(",")] if arg else [],
-    default=[],
-    help="comma-separated number of times to write each PART (default: once)",
-)
-parser.add_argument(
-    "--line",
-    action="store_true",
-    help="write the PARTs as a single line of llvm-strings output",
-)
-parser.add_argument(
-    "--offset",
-    type=int,
-    help="with --line, precede the line with OFFSET as printed by --radix=d",
-)
-args = parser.parse_intermixed_args()
-
-repeat = args.repeat or [1] * len(args.parts)
-if len(repeat) != len(args.parts):
-    parser.error("--repeat has %d counts for %d PARTs" % (len(repeat), len(args.parts)))
-
-out = b"".join(
-    (b"\0" * int(part) if part.isdigit() else part.encode()) * count
-    for part, count in zip(args.parts, repeat)
-)
-if args.line:
-    if args.offset is not None:
-        out = b"%7d " % args.offset + out
-    out += b"\n"
-# Write bytes, not text, so that "\n" is not written as "\r\n" on Windows.
-sys.stdout.buffer.write(out)

>From 3ade6c9672d60912b15b7ae5d6108bda7a6cf896 Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Wed, 23 Sep 2026 12:41:04 -0500
Subject: [PATCH 19/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 llvm/test/tools/llvm-strings/chunk-boundary.test | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/llvm/test/tools/llvm-strings/chunk-boundary.test b/llvm/test/tools/llvm-strings/chunk-boundary.test
index 658a7b1cabbb9d..8c0dee45a0a5d3 100644
--- a/llvm/test/tools/llvm-strings/chunk-boundary.test
+++ b/llvm/test/tools/llvm-strings/chunk-boundary.test
@@ -46,7 +46,7 @@
 # RUN: printf "%16383sS%.16384dE%4s" | tr " " "\0" > %t.6
 # RUN: printf "S%.16384dE\n" > %t.6.expected
 # RUN: llvm-strings %t.6 > %t.6.out
-# RUN: diff --strip-trailing-cr %t.6.expected %t.6.out
+# RUN: diff %t.6.expected %t.6.out
 
 ## Case 7: the minimum length is greater than the chunk size, so a candidate
 ## has to be buffered across more than one chunk before it is known to be long
@@ -55,7 +55,7 @@
 # RUN: printf "%8s%.20000d%4s" | tr " " "\0" > %t.7
 # RUN: printf "      8 %.20000d\n" > %t.7.expected
 # RUN: llvm-strings --radix=d --bytes=20000 %t.7 > %t.7.out
-# RUN: diff --strip-trailing-cr %t.7.expected %t.7.out
+# RUN: diff %t.7.expected %t.7.out
 
 ## Case 8: as case 7, but the run is one byte short of the minimum length, so
 ## the buffered candidate is discarded and nothing is printed.

>From 071aea81bb656b2576e233f5999445a02866e45d Mon Sep 17 00:00:00 2001
From: ShengYi Hung <aokblast at FreeBSD.org>
Date: Fri, 25 Sep 2026 15:16:29 -0500
Subject: [PATCH 20/20] fixup! [llvm-strings] Use small buffer instead of
 reading whole file

---
 llvm/test/tools/llvm-strings/chunk-boundary.test | 3 +++
 1 file changed, 3 insertions(+)

diff --git a/llvm/test/tools/llvm-strings/chunk-boundary.test b/llvm/test/tools/llvm-strings/chunk-boundary.test
index 8c0dee45a0a5d3..6977856d6eaded 100644
--- a/llvm/test/tools/llvm-strings/chunk-boundary.test
+++ b/llvm/test/tools/llvm-strings/chunk-boundary.test
@@ -5,6 +5,9 @@
 ## Case 1: at least min string size appears before the boundary, unprintable
 ## byte as first byte of the next chunk. The string is printed on its own,
 ## with the offset of its start (16376 = 16384 - 8).
+## Some older versions of printf on Windows produced \r\n instead of \n,
+## which would cause the subsequent diff to fail. Make sure you are using an
+## up-to-date version.
 # RUN: printf "%16376sENDCHUNK%4s" | tr " " "\0" > %t.1
 # RUN: llvm-strings --radix=d %t.1 | FileCheck %s --check-prefix=CASE1 --match-full-lines --strict-whitespace --implicit-check-not={{.}}
 



More information about the llvm-commits mailing list