[llvm] [llvm-strings] Use small buffer instead of reading whole file (PR #163073)
via llvm-commits
llvm-commits at lists.llvm.org
Sun Sep 13 14:19:48 PDT 2026
github-actions[bot] wrote:
<!--LLVM CODE FORMAT COMMENT: {clang-format}-->
:warning: C/C++ code formatter, clang-format found issues in your code. :warning:
<details>
<summary>
You can test this locally with the following command:
</summary>
``````````bash
git-clang-format --diff origin/main HEAD --extensions cpp -- llvm/tools/llvm-strings/llvm-strings.cpp --diff_from_common_commit
``````````
:warning:
The reproduction instructions above might return results for more than one PR
in a stack if you are using a stacked PR workflow. You can limit the results by
changing `origin/main` to the base branch/commit you want to compare against.
:warning:
</details>
<details>
<summary>
View the diff from clang-format here.
</summary>
``````````diff
diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 4109842bb..d46212a52 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -138,185 +138,186 @@ static void strings(raw_ostream &OS, StringRef FileName,
<< errorToErrorCode(ReadBytesOrErr.takeError()).message() << '\n';
return;
=======
- const char *B = Contents.begin();
- const char *P = nullptr, *E = nullptr, *S = nullptr;
- for (P = Contents.begin(), E = Contents.end(); P < E; ++P) {
- if (isPrint(*P) || *P == '\t') {
- if (S == nullptr)
- S = P;
- } else if (S) {
- Print(S - B, StringRef(S, P - S));
- S = nullptr;
+ const char *B = Contents.begin();
+ const char *P = nullptr, *E = nullptr, *S = nullptr;
+ for (P = Contents.begin(), E = Contents.end(); P < E; ++P) {
+ if (isPrint(*P) || *P == '\t') {
+ if (S == nullptr)
+ S = P;
+ } else if (S) {
+ Print(S - B, StringRef(S, P - S));
+ S = nullptr;
>>>>>>> main
- }
- size_t ChunkSize = *ReadBytesOrErr;
- if (ChunkSize == 0)
- break;
+ }
+ size_t ChunkSize = *ReadBytesOrErr;
+ if (ChunkSize == 0)
+ break;
- // To prevent performance regression under O0, access the raw pointer
- // instead of using methods provided by the standard library, which are not
- // inlined under O0.
- const char *const Begin = Buffer.data();
- const char *const End = Begin + ChunkSize;
- const char *Cur = Begin;
+ // To prevent performance regression under O0, access the raw pointer
+ // instead of using methods provided by the standard library, which are
+ // not inlined under O0.
+ const char *const Begin = Buffer.data();
+ const char *const End = Begin + ChunkSize;
+ const char *Cur = Begin;
- // Handle the remaining part from the previous chunk.
- // The previous chunk can be either shorter than MinSize or part of the
- // string.
- // Keep the buffer size bounded. With a small Min, a long string spanning
- // multiple chunks will have at most DefaultReadChunkSize bytes, since the
- // buffer is printed immediately with the header (guarded by the second if).
- // With a large Min, the buffer must hold at least Min bytes, since we need
- // enough data to decide whether to print it.
- if (InString || !Candidate.empty()) {
- // Find the end of the current string.
- while (Cur != End && isStringChar(*Cur))
- ++Cur;
- size_t Len = Cur - Begin;
- if (InString) {
- // Print the remaining part if the previous chunk has already printed
- // the header. E.g. header: aaaaa | bbbbb, where | is the chunk
- // boundary.
- // Output: Header: aaaaabbbbb, where bbbbb is printed in here.
- OS << StringRef(Begin, Len);
- } else if (Candidate.size() + Len >= Min) {
- // If the header hasn't been printed yet (e.g. the previous candidate
- // was smaller than Min), but we can print it now, print the header
- // first, followed by the candidate from the previous chunk and the
- // current string. E.g. aa | bbbbbb
- // Output Header: aabbbbbb, where aabbbbbb is printed in here.
- printHeader(ChunkOffset - Candidate.size());
- OS << Candidate << StringRef(Begin, Len);
- Candidate.clear();
- InString = true;
- } else if (Cur == End) {
- // If the current chunk + previous candidate is still smaller than Min,
- // append it to Candidate.
- Candidate.append(Begin, End);
- } else {
- // If the string has terminated but is still smaller than Min, clear the
- // buffer since it is too short to print.
- Candidate.clear();
- }
+ // Handle the remaining part from the previous chunk.
+ // The previous chunk can be either shorter than MinSize or part of the
+ // string.
+ // Keep the buffer size bounded. With a small Min, a long string
+ // spanning multiple chunks will have at most DefaultReadChunkSize
+ // bytes, since the buffer is printed immediately with the header
+ // (guarded by the second if). With a large Min, the buffer must hold at
+ // least Min bytes, since we need enough data to decide whether to print
+ // it.
+ if (InString || !Candidate.empty()) {
+ // Find the end of the current string.
+ while (Cur != End && isStringChar(*Cur))
+ ++Cur;
+ size_t Len = Cur - Begin;
+ if (InString) {
+ // Print the remaining part if the previous chunk has already
+ // printed the header. E.g. header: aaaaa | bbbbb, where | is the
+ // chunk boundary. Output: Header: aaaaabbbbb, where bbbbb is
+ // printed in here.
+ OS << StringRef(Begin, Len);
+ } else if (Candidate.size() + Len >= Min) {
+ // If the header hasn't been printed yet (e.g. the previous
+ // candidate was smaller than Min), but we can print it now, print
+ // the header first, followed by the candidate from the previous
+ // chunk and the current string. E.g. aa | bbbbbb Output Header:
+ // aabbbbbb, where aabbbbbb is printed in here.
+ printHeader(ChunkOffset - Candidate.size());
+ OS << Candidate << StringRef(Begin, Len);
+ Candidate.clear();
+ InString = true;
+ } else if (Cur == End) {
+ // If the current chunk + previous candidate is still smaller than
+ // Min, append it to Candidate.
+ Candidate.append(Begin, End);
+ } else {
+ // If the string has terminated but is still smaller than Min, clear
+ // the buffer since it is too short to print.
+ Candidate.clear();
+ }
+
+ if (Cur == End) {
+ // Finish handling the current chunk and update ChunkOffset.
+ ChunkOffset += ChunkSize;
+ continue;
+ }
+ if (InString) {
+ // We haven't reached the end of the chunk, which means the string
+ // is terminated. Add a '\n' to start printing a new string.
+ OS << '\n';
+ InString = false;
+ }
+ }
+
+ // At this point, we are always at the start of a new string because the
+ // remaining part of the previous string has already been handled.
+ const char *StrHead = nullptr;
+ for (; Cur != End; ++Cur) {
+ if (isStringChar(*Cur)) {
+ // Find the start of the next string.
+ if (!StrHead)
+ StrHead = Cur;
+ } else if (StrHead) {
+ // If it is not a printable character, we have reached the end of
+ // the current string. Print it if long enough.
+ if (static_cast<size_t>(Cur - StrHead) >= Min) {
+ printHeader(ChunkOffset + (StrHead - Begin));
+ OS << StringRef(StrHead, Cur - StrHead) << '\n';
+ }
+ StrHead = nullptr;
+ }
+ }
- if (Cur == End) {
- // Finish handling the current chunk and update ChunkOffset.
+ // The last string could span multiple chunks. If it is larger than Min,
+ // print the header immediately and set the InString flag to avoid
+ // printing it again.
+ if (StrHead) {
+ size_t Len = End - StrHead;
+ // Print it, or append it to Candidate if it is too short.
+ if (Len >= Min) {
+ printHeader(ChunkOffset + (StrHead - Begin));
+ OS << StringRef(StrHead, Len);
+ InString = true;
+ } else {
+ Candidate.append(StrHead, End);
+ }
+ }
ChunkOffset += ChunkSize;
- continue;
}
- if (InString) {
- // We haven't reached the end of the chunk, which means the string is
- // terminated. Add a '\n' to start printing a new string.
+
+ if (InString)
OS << '\n';
- InString = false;
- }
}
- // At this point, we are always at the start of a new string because the
- // remaining part of the previous string has already been handled.
- const char *StrHead = nullptr;
- for (; Cur != End; ++Cur) {
- if (isStringChar(*Cur)) {
- // Find the start of the next string.
- if (!StrHead)
- StrHead = Cur;
- } else if (StrHead) {
- // If it is not a printable character, we have reached the end of the
- // current string. Print it if long enough.
- if (static_cast<size_t>(Cur - StrHead) >= Min) {
- printHeader(ChunkOffset + (StrHead - Begin));
- OS << StringRef(StrHead, Cur - StrHead) << '\n';
- }
- StrHead = nullptr;
+ int main(int argc, char **argv) {
+ InitLLVM X(argc, argv);
+ BumpPtrAllocator A;
+ StringSaver Saver(A);
+ StringsOptTable Tbl;
+ ToolName = argv[0];
+ opt::InputArgList Args =
+ Tbl.parseArgs(argc, argv, OPT_UNKNOWN, Saver,
+ [&](StringRef Msg) { reportCmdLineError(Msg); });
+ if (Args.hasArg(OPT_help)) {
+ Tbl.printHelp(
+ outs(),
+ (Twine(ToolName) + " [options] <input object files>").str().c_str(),
+ "llvm string dumper");
+ // TODO Replace this with OptTable API once it adds extrahelp support.
+ outs() << "\nPass @FILE as argument to read options from FILE.\n";
+ return 0;
+ }
+ if (Args.hasArg(OPT_version)) {
+ outs() << ToolName << '\n';
+ cl::PrintVersionMessage();
+ return 0;
}
- }
- // The last string could span multiple chunks. If it is larger than Min,
- // print the header immediately and set the InString flag to avoid printing
- // it again.
- if (StrHead) {
- size_t Len = End - StrHead;
- // Print it, or append it to Candidate if it is too short.
- if (Len >= Min) {
- printHeader(ChunkOffset + (StrHead - Begin));
- OS << StringRef(StrHead, Len);
- InString = true;
+ parseIntArg(Args, OPT_bytes_EQ, MinLength);
+ PrintFileName = Args.hasArg(OPT_print_file_name);
+ Arg *RadixArg = Args.getLastArg(OPT_radix_EQ);
+ if (!RadixArg) {
+ Radix = Radix::None;
} else {
- Candidate.append(StrHead, End);
+ Radix = llvm::StringSwitch<enum Radix>(RadixArg->getValue())
+ .Case("o", Radix::Octal)
+ .Case("d", Radix::Decimal)
+ .Case("x", Radix::Hexadecimal)
+ .Default(Radix::None);
+ if (Radix == Radix::None)
+ reportCmdLineError("'" + StringRef(RadixArg->getValue()) +
+ "' is not a valid value for '" +
+ RadixArg->getSpelling() + "'");
}
- }
- ChunkOffset += ChunkSize;
- }
-
- if (InString)
- OS << '\n';
-}
-
-int main(int argc, char **argv) {
- InitLLVM X(argc, argv);
- BumpPtrAllocator A;
- StringSaver Saver(A);
- StringsOptTable Tbl;
- ToolName = argv[0];
- opt::InputArgList Args =
- Tbl.parseArgs(argc, argv, OPT_UNKNOWN, Saver,
- [&](StringRef Msg) { reportCmdLineError(Msg); });
- if (Args.hasArg(OPT_help)) {
- Tbl.printHelp(
- outs(),
- (Twine(ToolName) + " [options] <input object files>").str().c_str(),
- "llvm string dumper");
- // TODO Replace this with OptTable API once it adds extrahelp support.
- outs() << "\nPass @FILE as argument to read options from FILE.\n";
- return 0;
- }
- if (Args.hasArg(OPT_version)) {
- outs() << ToolName << '\n';
- cl::PrintVersionMessage();
- return 0;
- }
-
- parseIntArg(Args, OPT_bytes_EQ, MinLength);
- PrintFileName = Args.hasArg(OPT_print_file_name);
- Arg *RadixArg = Args.getLastArg(OPT_radix_EQ);
- if (!RadixArg) {
- Radix = Radix::None;
- } else {
- Radix = llvm::StringSwitch<enum Radix>(RadixArg->getValue())
- .Case("o", Radix::Octal)
- .Case("d", Radix::Decimal)
- .Case("x", Radix::Hexadecimal)
- .Default(Radix::None);
- if (Radix == Radix::None)
- reportCmdLineError("'" + StringRef(RadixArg->getValue()) +
- "' is not a valid value for '" +
- RadixArg->getSpelling() + "'");
- }
- if (MinLength == 0) {
- errs() << "invalid minimum string length 0\n";
- return EXIT_FAILURE;
- }
+ if (MinLength == 0) {
+ errs() << "invalid minimum string length 0\n";
+ return EXIT_FAILURE;
+ }
- std::vector<std::string> InputFileNames = Args.getAllArgValues(OPT_INPUT);
- if (InputFileNames.empty())
- InputFileNames.push_back("-");
+ std::vector<std::string> InputFileNames = Args.getAllArgValues(OPT_INPUT);
+ if (InputFileNames.empty())
+ InputFileNames.push_back("-");
- for (const auto &File : InputFileNames) {
- if (File == "-") {
- strings(llvm::outs(), "{standard input}", sys::fs::getStdinHandle());
- } else {
- Expected<sys::fs::file_t> FDOrErr =
- sys::fs::openNativeFileForRead(File, sys::fs::OF_TextWithCRLF);
- if (!FDOrErr) {
- errs() << File
- << ": cannot open file: " << toString(FDOrErr.takeError())
- << '\n';
- continue;
+ for (const auto &File : InputFileNames) {
+ if (File == "-") {
+ strings(llvm::outs(), "{standard input}", sys::fs::getStdinHandle());
+ } else {
+ Expected<sys::fs::file_t> FDOrErr =
+ sys::fs::openNativeFileForRead(File, sys::fs::OF_TextWithCRLF);
+ if (!FDOrErr) {
+ errs() << File
+ << ": cannot open file: " << toString(FDOrErr.takeError())
+ << '\n';
+ continue;
+ }
+ strings(llvm::outs(), File, *FDOrErr);
+ }
}
- strings(llvm::outs(), File, *FDOrErr);
- }
- }
- return EXIT_SUCCESS;
-}
+ return EXIT_SUCCESS;
+ }
``````````
</details>
https://github.com/llvm/llvm-project/pull/163073
More information about the llvm-commits
mailing list