[llvm] [llvm-strings] Add --encoding={s,S,utf8} option (PR #221794)

James Henderson via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 01:51:07 PDT 2026


================
@@ -146,87 +288,117 @@ static void strings(raw_ostream &OS, StringRef FileName,
     // With a large Min, the buffer must hold at least Min bytes, since we need
     // enough data to decide whether to print it.
     if (InString || !Candidate.empty()) {
+      const char *StringEnd;
+      std::mbstate_t PrevMBState;
+      size_t Len = 0;
+      bool EndOfChunk;
+
       // Find the end of the current string.
-      while (Cur != End && isStringChar(*Cur))
-        ++Cur;
-      size_t Len = Cur - Begin;
+      for (;;) {
+        StringEnd = Cur;
+        PrevMBState = MBState;
+        EndOfChunk = !Read(Cur, End, MBState, AtEOF, Ch);
+        if (EndOfChunk || !IsStringChar(Ch))
+          break;
+        ++Len;
+      }
+
+      size_t Size = StringEnd - Begin;
       if (InString) {
         // Print the remaining part if the previous chunk has already printed
         // the header. E.g. header: aaaaa | bbbbb, where | is the chunk
         // boundary.
         // Output: Header: aaaaabbbbb, where bbbbb is printed in here.
-        OS << StringRef(Begin, Len);
-      } else if (Candidate.size() + Len >= Min) {
-        // If the header hasn't been printed yet (e.g. the previous candidate
+        OS << StringRef(Begin, Size);
+      } else if (CandidateLength + Len >= Min) {
+        // If the header hasn't been printed yet (i.e. the previous candidate
         // was smaller than Min), but we can print it now, print the header
         // first, followed by the candidate from the previous chunk and the
         // current string. E.g. aa | bbbbbb
         // Output Header: aabbbbbb, where aabbbbbb is printed in here.
         PrintHeader(ChunkOffset - Candidate.size());
-        OS << Candidate << StringRef(Begin, Len);
+        OS << Candidate << StringRef(Begin, Size);
         Candidate.clear();
+        CandidateLength = 0;
         InString = true;
-      } else if (Cur == End) {
+      } else if (EndOfChunk) {
         // If the current chunk + previous candidate is still smaller than Min,
         // append it to Candidate.
-        Candidate.append(Begin, End);
+        Candidate.append(Begin, StringEnd);
+        CandidateLength += Len;
       } else {
         // If the string has terminated but is still smaller than Min, clear the
         // buffer since it is too short to print.
         Candidate.clear();
+        CandidateLength = 0;
       }
 
-      if (Cur == End) {
+      if (EndOfChunk || Cur == End) {
         // Finish handling the current chunk and update ChunkOffset.
-        ChunkOffset += ChunkSize;
+        ChunkOffset += Cur - Begin;
+        Buffer.erase(Buffer.begin(), Cur);
         continue;
       }
+
       if (InString) {
         // We haven't reached the end of the chunk, which means the string is
         // terminated. Add a '\n' to start printing a new string.
-        OS << '\n';
+        EndString(PrevMBState);
         InString = false;
       }
     }
 
     // At this point, we are always at the start of a new string because the
     // remaining part of the previous string has already been handled.
     const char *StrHead = nullptr;
-    for (; Cur != End; ++Cur) {
-      if (isStringChar(*Cur)) {
+    size_t Len = 0;
----------------
jh7370 wrote:

Nit: Given that we have both string and character lengths relevant to the code now, it might be worth naming this `StrLen`, to avoid any ambiguity.

https://github.com/llvm/llvm-project/pull/221794


More information about the llvm-commits mailing list