[llvm] [llvm-strings] Add --encoding={s,S,utf8} option (PR #221794)
James Henderson via llvm-commits
llvm-commits at lists.llvm.org
Fri Oct 2 01:51:07 PDT 2026
================
@@ -101,6 +111,127 @@ static void strings(raw_ostream &OS, StringRef FileName,
}
};
+ std::locale Loc("");
+ auto &Cvt = std::use_facet<std::codecvt<wchar_t, char, std::mbstate_t>>(Loc);
+ auto &Ctype = std::use_facet<std::ctype<wchar_t>>(Loc);
+
+ auto IsStringChar = [&Ctype](UTF32 Ch) {
+ if (Ch == '\t')
+ return true;
+
+ switch (Encoding) {
+ case Encoding::Ascii:
+ return isPrint(Ch);
+
+ case Encoding::Locale:
+ return Ctype.is(std::ctype_base::print, Ch);
+
+ case Encoding::Utf8:
+ return sys::unicode::isPrintable(Ch);
+ }
+
+ llvm_unreachable("unhandled encoding");
+ };
+
+ // Tries to read one character from the bytes P...E and store it in Ch.
+ // Returns true if a (possibly invalid) character was read, false otherwise.
+ //
+ // If P...E starts with a valid complete character, P and MBState are updated
+ // and true is returned.
+ // If P...E is empty, or holds an incomplete character and AtEOF is false, P
+ // and MBState are unchanged, Ch is set to 0, and false is returned. This is
+ // intended to allow more bytes to be read and Read to be called again.
+ // If P...E holds an incomplete character and AtEOF is true, or if P...E
+ // starts with an invalid character, the first byte is skipped and MBState is
+ // reset to allow continuing from the next point, Ch is set to 0, and true is
+ // returned.
+ auto Read = [&Cvt](const char *&P, const char *E, std::mbstate_t &MBState,
+ bool AtEOF, UTF32 &Ch) -> bool {
+ if (P == E)
+ return false;
+
+ switch (Encoding) {
+ case Encoding::Ascii:
+ Ch = *P++;
+ return true;
+
+ case Encoding::Locale: {
+ const char *N;
+ wchar_t WCh;
+ wchar_t *WNext;
+ std::mbstate_t SaveMBState = MBState;
+ const auto Res = Cvt.in(MBState, P, E, N, &WCh, &WCh + 1, WNext);
+ assert(Res != std::codecvt_base::noconv);
+
+ if (WNext != &WCh) {
+ // Only treat a non-null character as a successful conversion, as a
+ // null character may be the result of an incomplete multibyte
+ // character followed by a null byte.
+ if (WCh) {
+ // Note: this assumes wchar_t is UCS2 or UTF32.
+ Ch = WCh;
+ P = N;
+ return true;
+ }
+ // Otherwise treat it as an error. A null byte is safe to treat as an
+ // error, as a null byte is never printable in any locale.
+ } else if ((Res == std::codecvt_base::ok ||
+ Res == std::codecvt_base::partial) &&
+ !AtEOF) {
+ // If we got a partial result but no character was written, we have an
+ // incomplete multibyte character. Do not treat this as an error,
+ // instead reset the conversion state so that we can try again if/when
+ // we have more characters, unless we know there are no more characters.
----------------
jh7370 wrote:
Should "characters" be "bytes" in both instances in this line?
https://github.com/llvm/llvm-project/pull/221794
More information about the llvm-commits
mailing list