[llvm] [llvm-strings] Add --encoding={s,S,utf8} option (PR #221794)

via llvm-commits llvm-commits at lists.llvm.org
Mon Sep 28 21:49:50 PDT 2026


================
@@ -101,6 +111,125 @@ static void strings(raw_ostream &OS, StringRef FileName,
     }
   };
 
+  std::locale Loc("");
+  auto &Cvt = std::use_facet<std::codecvt<wchar_t, char, std::mbstate_t>>(Loc);
+  auto &Ctype = std::use_facet<std::ctype<wchar_t>>(Loc);
+
+  auto IsStringChar = [&Ctype](UTF32 Ch) {
+    if (Ch == '\t')
+      return true;
+
+    switch (Encoding) {
+    case Encoding::Ascii:
+      return isPrint(Ch);
+
+    case Encoding::Locale:
+      return Ctype.is(std::ctype_base::print, Ch);
+
+    case Encoding::Utf8:
+      return sys::unicode::isPrintable(Ch);
+    }
+
+    llvm_unreachable("unhandled encoding");
+  };
+
+  // Tries to read one character from the bytes P...E and store it in Ch.
+  // Returns true if a (possibly invalid) character was read, false otherwise.
+  //
+  // If P...E starts with a valid complete character, P and MBState are updated
+  // and true is returned.
+  // If P...E is empty, or holds an incomplete character and AtEOF is false, P
+  // and MBState are unchanged, Ch is set to 0, and false is returned. This is
+  // intended to allow more bytes to be read and Read to be called again.
+  // If P...E holds an incomplete character and AtEOF is true, or if P...E
+  // starts with an invalid character, the first byte is skipped and MBState is
+  // reset to allow continuing from the next point, Ch is set to 0, and true is
+  // returned.
+  auto Read = [&Cvt](const char *&P, const char *E, std::mbstate_t &MBState,
+                     bool AtEOF, UTF32 &Ch) -> bool {
+    if (P == E)
+      return false;
+
+    switch (Encoding) {
+    case Encoding::Ascii:
+      Ch = *P++;
+      return true;
+
+    case Encoding::Locale: {
+      const char *N;
+      wchar_t WCh;
+      wchar_t *WNext;
+      std::mbstate_t SaveMBState = MBState;
+      const auto Res = Cvt.in(MBState, P, E, N, &WCh, &WCh + 1, WNext);
+      assert(Res != std::codecvt_base::noconv);
+
+      if (WNext != &WCh) {
+        // Only treat a non-null character as a successful conversion, as a
+        // null character may be the result of an incomplete multibyte
+        // character followed by a null byte.
+        if (WCh) {
+          // Note: this assumes wchar_t is UCS2 or UTF32.
+          Ch = WCh;
+          P = N;
+          return true;
+        }
+        // Otherwise treat it as an error. A null byte is safe to treat as an
+        // error, as a null byte is never printable in any locale.
+      } else if (Res == std::codecvt_base::partial && !AtEOF) {
+        // If we got a partial result but no character was written, we have an
+        // incomplete multibyte character.  Do not treat this as an error,
+        // instead reset the conversion state so that we can try again if/when
+        // we have more characters, unless we know there are no more characters.
+        Ch = 0;
+        MBState = SaveMBState;
+        return false;
+      }
+      // If there was any error, reset the state to allow the next byte to
+      // start a character.
+      Ch = 0;
+      MBState = {};
+      ++P;
+      return true;
+    }
+
+    case Encoding::Utf8: {
+      const UTF8 *UP = reinterpret_cast<const UTF8 *>(P);
+      const UTF8 *UE = reinterpret_cast<const UTF8 *>(E);
+      UTF32 *Next = &Ch;
+      const auto Res =
+          ConvertUTF8toUTF32Partial(&UP, UE, &Next, &Ch + 1, strictConversion);
+      if (Next != &Ch) {
+        if (Ch) {
+          P = reinterpret_cast<const char *>(UP);
+          return true;
+        }
+      } else if (Res == sourceExhausted) {
----------------
aokblast wrote:

I think we will spin over here without checking EOF?

https://github.com/llvm/llvm-project/pull/221794


More information about the llvm-commits mailing list