[llvm] [llvm-strings] Use small buffer instead of reading whole file (PR #163073)

via llvm-commits llvm-commits at lists.llvm.org
Sun Sep 13 14:19:48 PDT 2026


github-actions[bot] wrote:

<!--LLVM CODE FORMAT COMMENT: {clang-format}-->


:warning: C/C++ code formatter, clang-format found issues in your code. :warning:

<details>
<summary>
You can test this locally with the following command:
</summary>

``````````bash
git-clang-format --diff origin/main HEAD --extensions cpp -- llvm/tools/llvm-strings/llvm-strings.cpp --diff_from_common_commit
``````````

:warning:
The reproduction instructions above might return results for more than one PR
in a stack if you are using a stacked PR workflow. You can limit the results by
changing `origin/main` to the base branch/commit you want to compare against.
:warning:

</details>

<details>
<summary>
View the diff from clang-format here.
</summary>

``````````diff
diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 4109842bb..d46212a52 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -138,185 +138,186 @@ static void strings(raw_ostream &OS, StringRef FileName,
              << errorToErrorCode(ReadBytesOrErr.takeError()).message() << '\n';
       return;
 =======
-  const char *B = Contents.begin();
-  const char *P = nullptr, *E = nullptr, *S = nullptr;
-  for (P = Contents.begin(), E = Contents.end(); P < E; ++P) {
-    if (isPrint(*P) || *P == '\t') {
-      if (S == nullptr)
-        S = P;
-    } else if (S) {
-      Print(S - B, StringRef(S, P - S));
-      S = nullptr;
+      const char *B = Contents.begin();
+      const char *P = nullptr, *E = nullptr, *S = nullptr;
+      for (P = Contents.begin(), E = Contents.end(); P < E; ++P) {
+        if (isPrint(*P) || *P == '\t') {
+          if (S == nullptr)
+            S = P;
+        } else if (S) {
+          Print(S - B, StringRef(S, P - S));
+          S = nullptr;
 >>>>>>> main
-    }
-    size_t ChunkSize = *ReadBytesOrErr;
-    if (ChunkSize == 0)
-      break;
+        }
+        size_t ChunkSize = *ReadBytesOrErr;
+        if (ChunkSize == 0)
+          break;
 
-    // To prevent performance regression under O0, access the raw pointer
-    // instead of using methods provided by the standard library, which are not
-    // inlined under O0.
-    const char *const Begin = Buffer.data();
-    const char *const End = Begin + ChunkSize;
-    const char *Cur = Begin;
+        // To prevent performance regression under O0, access the raw pointer
+        // instead of using methods provided by the standard library, which are
+        // not inlined under O0.
+        const char *const Begin = Buffer.data();
+        const char *const End = Begin + ChunkSize;
+        const char *Cur = Begin;
 
-    // Handle the remaining part from the previous chunk.
-    // The previous chunk can be either shorter than MinSize or part of the
-    // string.
-    // Keep the buffer size bounded. With a small Min, a long string spanning
-    // multiple chunks will have at most DefaultReadChunkSize bytes, since the
-    // buffer is printed immediately with the header (guarded by the second if).
-    // With a large Min, the buffer must hold at least Min bytes, since we need
-    // enough data to decide whether to print it.
-    if (InString || !Candidate.empty()) {
-      // Find the end of the current string.
-      while (Cur != End && isStringChar(*Cur))
-        ++Cur;
-      size_t Len = Cur - Begin;
-      if (InString) {
-        // Print the remaining part if the previous chunk has already printed
-        // the header. E.g. header: aaaaa | bbbbb, where | is the chunk
-        // boundary.
-        // Output: Header: aaaaabbbbb, where bbbbb is printed in here.
-        OS << StringRef(Begin, Len);
-      } else if (Candidate.size() + Len >= Min) {
-        // If the header hasn't been printed yet (e.g. the previous candidate
-        // was smaller than Min), but we can print it now, print the header
-        // first, followed by the candidate from the previous chunk and the
-        // current string. E.g. aa | bbbbbb
-        // Output Header: aabbbbbb, where aabbbbbb is printed in here.
-        printHeader(ChunkOffset - Candidate.size());
-        OS << Candidate << StringRef(Begin, Len);
-        Candidate.clear();
-        InString = true;
-      } else if (Cur == End) {
-        // If the current chunk + previous candidate is still smaller than Min,
-        // append it to Candidate.
-        Candidate.append(Begin, End);
-      } else {
-        // If the string has terminated but is still smaller than Min, clear the
-        // buffer since it is too short to print.
-        Candidate.clear();
-      }
+        // Handle the remaining part from the previous chunk.
+        // The previous chunk can be either shorter than MinSize or part of the
+        // string.
+        // Keep the buffer size bounded. With a small Min, a long string
+        // spanning multiple chunks will have at most DefaultReadChunkSize
+        // bytes, since the buffer is printed immediately with the header
+        // (guarded by the second if). With a large Min, the buffer must hold at
+        // least Min bytes, since we need enough data to decide whether to print
+        // it.
+        if (InString || !Candidate.empty()) {
+          // Find the end of the current string.
+          while (Cur != End && isStringChar(*Cur))
+            ++Cur;
+          size_t Len = Cur - Begin;
+          if (InString) {
+            // Print the remaining part if the previous chunk has already
+            // printed the header. E.g. header: aaaaa | bbbbb, where | is the
+            // chunk boundary. Output: Header: aaaaabbbbb, where bbbbb is
+            // printed in here.
+            OS << StringRef(Begin, Len);
+          } else if (Candidate.size() + Len >= Min) {
+            // If the header hasn't been printed yet (e.g. the previous
+            // candidate was smaller than Min), but we can print it now, print
+            // the header first, followed by the candidate from the previous
+            // chunk and the current string. E.g. aa | bbbbbb Output Header:
+            // aabbbbbb, where aabbbbbb is printed in here.
+            printHeader(ChunkOffset - Candidate.size());
+            OS << Candidate << StringRef(Begin, Len);
+            Candidate.clear();
+            InString = true;
+          } else if (Cur == End) {
+            // If the current chunk + previous candidate is still smaller than
+            // Min, append it to Candidate.
+            Candidate.append(Begin, End);
+          } else {
+            // If the string has terminated but is still smaller than Min, clear
+            // the buffer since it is too short to print.
+            Candidate.clear();
+          }
+
+          if (Cur == End) {
+            // Finish handling the current chunk and update ChunkOffset.
+            ChunkOffset += ChunkSize;
+            continue;
+          }
+          if (InString) {
+            // We haven't reached the end of the chunk, which means the string
+            // is terminated. Add a '\n' to start printing a new string.
+            OS << '\n';
+            InString = false;
+          }
+        }
+
+        // At this point, we are always at the start of a new string because the
+        // remaining part of the previous string has already been handled.
+        const char *StrHead = nullptr;
+        for (; Cur != End; ++Cur) {
+          if (isStringChar(*Cur)) {
+            // Find the start of the next string.
+            if (!StrHead)
+              StrHead = Cur;
+          } else if (StrHead) {
+            // If it is not a printable character, we have reached the end of
+            // the current string. Print it if long enough.
+            if (static_cast<size_t>(Cur - StrHead) >= Min) {
+              printHeader(ChunkOffset + (StrHead - Begin));
+              OS << StringRef(StrHead, Cur - StrHead) << '\n';
+            }
+            StrHead = nullptr;
+          }
+        }
 
-      if (Cur == End) {
-        // Finish handling the current chunk and update ChunkOffset.
+        // The last string could span multiple chunks. If it is larger than Min,
+        // print the header immediately and set the InString flag to avoid
+        // printing it again.
+        if (StrHead) {
+          size_t Len = End - StrHead;
+          // Print it, or append it to Candidate if it is too short.
+          if (Len >= Min) {
+            printHeader(ChunkOffset + (StrHead - Begin));
+            OS << StringRef(StrHead, Len);
+            InString = true;
+          } else {
+            Candidate.append(StrHead, End);
+          }
+        }
         ChunkOffset += ChunkSize;
-        continue;
       }
-      if (InString) {
-        // We haven't reached the end of the chunk, which means the string is
-        // terminated. Add a '\n' to start printing a new string.
+
+      if (InString)
         OS << '\n';
-        InString = false;
-      }
     }
 
-    // At this point, we are always at the start of a new string because the
-    // remaining part of the previous string has already been handled.
-    const char *StrHead = nullptr;
-    for (; Cur != End; ++Cur) {
-      if (isStringChar(*Cur)) {
-        // Find the start of the next string.
-        if (!StrHead)
-          StrHead = Cur;
-      } else if (StrHead) {
-        // If it is not a printable character, we have reached the end of the
-        // current string. Print it if long enough.
-        if (static_cast<size_t>(Cur - StrHead) >= Min) {
-          printHeader(ChunkOffset + (StrHead - Begin));
-          OS << StringRef(StrHead, Cur - StrHead) << '\n';
-        }
-        StrHead = nullptr;
+    int main(int argc, char **argv) {
+      InitLLVM X(argc, argv);
+      BumpPtrAllocator A;
+      StringSaver Saver(A);
+      StringsOptTable Tbl;
+      ToolName = argv[0];
+      opt::InputArgList Args =
+          Tbl.parseArgs(argc, argv, OPT_UNKNOWN, Saver,
+                        [&](StringRef Msg) { reportCmdLineError(Msg); });
+      if (Args.hasArg(OPT_help)) {
+        Tbl.printHelp(
+            outs(),
+            (Twine(ToolName) + " [options] <input object files>").str().c_str(),
+            "llvm string dumper");
+        // TODO Replace this with OptTable API once it adds extrahelp support.
+        outs() << "\nPass @FILE as argument to read options from FILE.\n";
+        return 0;
+      }
+      if (Args.hasArg(OPT_version)) {
+        outs() << ToolName << '\n';
+        cl::PrintVersionMessage();
+        return 0;
       }
-    }
 
-    // The last string could span multiple chunks. If it is larger than Min,
-    // print the header immediately and set the InString flag to avoid printing
-    // it again.
-    if (StrHead) {
-      size_t Len = End - StrHead;
-      // Print it, or append it to Candidate if it is too short.
-      if (Len >= Min) {
-        printHeader(ChunkOffset + (StrHead - Begin));
-        OS << StringRef(StrHead, Len);
-        InString = true;
+      parseIntArg(Args, OPT_bytes_EQ, MinLength);
+      PrintFileName = Args.hasArg(OPT_print_file_name);
+      Arg *RadixArg = Args.getLastArg(OPT_radix_EQ);
+      if (!RadixArg) {
+        Radix = Radix::None;
       } else {
-        Candidate.append(StrHead, End);
+        Radix = llvm::StringSwitch<enum Radix>(RadixArg->getValue())
+                    .Case("o", Radix::Octal)
+                    .Case("d", Radix::Decimal)
+                    .Case("x", Radix::Hexadecimal)
+                    .Default(Radix::None);
+        if (Radix == Radix::None)
+          reportCmdLineError("'" + StringRef(RadixArg->getValue()) +
+                             "' is not a valid value for '" +
+                             RadixArg->getSpelling() + "'");
       }
-    }
-    ChunkOffset += ChunkSize;
-  }
-
-  if (InString)
-    OS << '\n';
-}
-
-int main(int argc, char **argv) {
-  InitLLVM X(argc, argv);
-  BumpPtrAllocator A;
-  StringSaver Saver(A);
-  StringsOptTable Tbl;
-  ToolName = argv[0];
-  opt::InputArgList Args =
-      Tbl.parseArgs(argc, argv, OPT_UNKNOWN, Saver,
-                    [&](StringRef Msg) { reportCmdLineError(Msg); });
-  if (Args.hasArg(OPT_help)) {
-    Tbl.printHelp(
-        outs(),
-        (Twine(ToolName) + " [options] <input object files>").str().c_str(),
-        "llvm string dumper");
-    // TODO Replace this with OptTable API once it adds extrahelp support.
-    outs() << "\nPass @FILE as argument to read options from FILE.\n";
-    return 0;
-  }
-  if (Args.hasArg(OPT_version)) {
-    outs() << ToolName << '\n';
-    cl::PrintVersionMessage();
-    return 0;
-  }
-
-  parseIntArg(Args, OPT_bytes_EQ, MinLength);
-  PrintFileName = Args.hasArg(OPT_print_file_name);
-  Arg *RadixArg = Args.getLastArg(OPT_radix_EQ);
-  if (!RadixArg) {
-    Radix = Radix::None;
-  } else {
-    Radix = llvm::StringSwitch<enum Radix>(RadixArg->getValue())
-                .Case("o", Radix::Octal)
-                .Case("d", Radix::Decimal)
-                .Case("x", Radix::Hexadecimal)
-                .Default(Radix::None);
-    if (Radix == Radix::None)
-      reportCmdLineError("'" + StringRef(RadixArg->getValue()) +
-                         "' is not a valid value for '" +
-                         RadixArg->getSpelling() + "'");
-  }
 
-  if (MinLength == 0) {
-    errs() << "invalid minimum string length 0\n";
-    return EXIT_FAILURE;
-  }
+      if (MinLength == 0) {
+        errs() << "invalid minimum string length 0\n";
+        return EXIT_FAILURE;
+      }
 
-  std::vector<std::string> InputFileNames = Args.getAllArgValues(OPT_INPUT);
-  if (InputFileNames.empty())
-    InputFileNames.push_back("-");
+      std::vector<std::string> InputFileNames = Args.getAllArgValues(OPT_INPUT);
+      if (InputFileNames.empty())
+        InputFileNames.push_back("-");
 
-  for (const auto &File : InputFileNames) {
-    if (File == "-") {
-      strings(llvm::outs(), "{standard input}", sys::fs::getStdinHandle());
-    } else {
-      Expected<sys::fs::file_t> FDOrErr =
-          sys::fs::openNativeFileForRead(File, sys::fs::OF_TextWithCRLF);
-      if (!FDOrErr) {
-        errs() << File
-               << ": cannot open file: " << toString(FDOrErr.takeError())
-               << '\n';
-        continue;
+      for (const auto &File : InputFileNames) {
+        if (File == "-") {
+          strings(llvm::outs(), "{standard input}", sys::fs::getStdinHandle());
+        } else {
+          Expected<sys::fs::file_t> FDOrErr =
+              sys::fs::openNativeFileForRead(File, sys::fs::OF_TextWithCRLF);
+          if (!FDOrErr) {
+            errs() << File
+                   << ": cannot open file: " << toString(FDOrErr.takeError())
+                   << '\n';
+            continue;
+          }
+          strings(llvm::outs(), File, *FDOrErr);
+        }
       }
-      strings(llvm::outs(), File, *FDOrErr);
-    }
-  }
 
-  return EXIT_SUCCESS;
-}
+      return EXIT_SUCCESS;
+    }

``````````

</details>


https://github.com/llvm/llvm-project/pull/163073


More information about the llvm-commits mailing list