[llvm] [llvm-strings] Add --encoding={s,S,utf8} option (PR #221794)

James Henderson via llvm-commits llvm-commits at lists.llvm.org
Fri Oct 2 01:51:07 PDT 2026


================
@@ -101,6 +111,127 @@ static void strings(raw_ostream &OS, StringRef FileName,
     }
   };
 
+  std::locale Loc("");
+  auto &Cvt = std::use_facet<std::codecvt<wchar_t, char, std::mbstate_t>>(Loc);
+  auto &Ctype = std::use_facet<std::ctype<wchar_t>>(Loc);
+
+  auto IsStringChar = [&Ctype](UTF32 Ch) {
+    if (Ch == '\t')
+      return true;
+
+    switch (Encoding) {
+    case Encoding::Ascii:
+      return isPrint(Ch);
+
+    case Encoding::Locale:
+      return Ctype.is(std::ctype_base::print, Ch);
+
+    case Encoding::Utf8:
+      return sys::unicode::isPrintable(Ch);
+    }
+
+    llvm_unreachable("unhandled encoding");
+  };
+
+  // Tries to read one character from the bytes P...E and store it in Ch.
+  // Returns true if a (possibly invalid) character was read, false otherwise.
+  //
+  // If P...E starts with a valid complete character, P and MBState are updated
+  // and true is returned.
+  // If P...E is empty, or holds an incomplete character and AtEOF is false, P
+  // and MBState are unchanged, Ch is set to 0, and false is returned. This is
+  // intended to allow more bytes to be read and Read to be called again.
+  // If P...E holds an incomplete character and AtEOF is true, or if P...E
+  // starts with an invalid character, the first byte is skipped and MBState is
+  // reset to allow continuing from the next point, Ch is set to 0, and true is
+  // returned.
+  auto Read = [&Cvt](const char *&P, const char *E, std::mbstate_t &MBState,
+                     bool AtEOF, UTF32 &Ch) -> bool {
+    if (P == E)
+      return false;
+
+    switch (Encoding) {
+    case Encoding::Ascii:
+      Ch = *P++;
+      return true;
+
+    case Encoding::Locale: {
+      const char *N;
+      wchar_t WCh;
+      wchar_t *WNext;
+      std::mbstate_t SaveMBState = MBState;
+      const auto Res = Cvt.in(MBState, P, E, N, &WCh, &WCh + 1, WNext);
+      assert(Res != std::codecvt_base::noconv);
+
+      if (WNext != &WCh) {
+        // Only treat a non-null character as a successful conversion, as a
+        // null character may be the result of an incomplete multibyte
+        // character followed by a null byte.
+        if (WCh) {
+          // Note: this assumes wchar_t is UCS2 or UTF32.
+          Ch = WCh;
+          P = N;
+          return true;
+        }
+        // Otherwise treat it as an error. A null byte is safe to treat as an
+        // error, as a null byte is never printable in any locale.
+      } else if ((Res == std::codecvt_base::ok ||
+                  Res == std::codecvt_base::partial) &&
+                 !AtEOF) {
+        // If we got a partial result but no character was written, we have an
+        // incomplete multibyte character.  Do not treat this as an error,
+        // instead reset the conversion state so that we can try again if/when
+        // we have more characters, unless we know there are no more characters.
----------------
jh7370 wrote:

Should "characters" be "bytes" in both instances in this line?

https://github.com/llvm/llvm-project/pull/221794


More information about the llvm-commits mailing list