[llvm] [llvm-strings] Add --encoding={s,S,utf8} option (PR #221794)
via llvm-commits
llvm-commits at lists.llvm.org
Mon Sep 28 21:51:16 PDT 2026
================
@@ -101,6 +111,125 @@ static void strings(raw_ostream &OS, StringRef FileName,
}
};
+ std::locale Loc("");
+ auto &Cvt = std::use_facet<std::codecvt<wchar_t, char, std::mbstate_t>>(Loc);
+ auto &Ctype = std::use_facet<std::ctype<wchar_t>>(Loc);
+
+ auto IsStringChar = [&Ctype](UTF32 Ch) {
+ if (Ch == '\t')
+ return true;
+
+ switch (Encoding) {
+ case Encoding::Ascii:
+ return isPrint(Ch);
+
+ case Encoding::Locale:
+ return Ctype.is(std::ctype_base::print, Ch);
+
+ case Encoding::Utf8:
+ return sys::unicode::isPrintable(Ch);
+ }
+
+ llvm_unreachable("unhandled encoding");
+ };
+
+ // Tries to read one character from the bytes P...E and store it in Ch.
+ // Returns true if a (possibly invalid) character was read, false otherwise.
+ //
+ // If P...E starts with a valid complete character, P and MBState are updated
+ // and true is returned.
+ // If P...E is empty, or holds an incomplete character and AtEOF is false, P
+ // and MBState are unchanged, Ch is set to 0, and false is returned. This is
+ // intended to allow more bytes to be read and Read to be called again.
+ // If P...E holds an incomplete character and AtEOF is true, or if P...E
+ // starts with an invalid character, the first byte is skipped and MBState is
+ // reset to allow continuing from the next point, Ch is set to 0, and true is
+ // returned.
+ auto Read = [&Cvt](const char *&P, const char *E, std::mbstate_t &MBState,
+ bool AtEOF, UTF32 &Ch) -> bool {
+ if (P == E)
+ return false;
+
+ switch (Encoding) {
+ case Encoding::Ascii:
+ Ch = *P++;
+ return true;
+
+ case Encoding::Locale: {
+ const char *N;
+ wchar_t WCh;
+ wchar_t *WNext;
+ std::mbstate_t SaveMBState = MBState;
+ const auto Res = Cvt.in(MBState, P, E, N, &WCh, &WCh + 1, WNext);
+ assert(Res != std::codecvt_base::noconv);
+
+ if (WNext != &WCh) {
+ // Only treat a non-null character as a successful conversion, as a
+ // null character may be the result of an incomplete multibyte
+ // character followed by a null byte.
+ if (WCh) {
+ // Note: this assumes wchar_t is UCS2 or UTF32.
+ Ch = WCh;
+ P = N;
+ return true;
+ }
+ // Otherwise treat it as an error. A null byte is safe to treat as an
+ // error, as a null byte is never printable in any locale.
+ } else if (Res == std::codecvt_base::partial && !AtEOF) {
----------------
aokblast wrote:
For some note: Actually, none of libstdcxx and libcxx goes here. They skip without returning partial in this case. But it is not your problem. It is a openquestion to the behavior.
https://github.com/llvm/llvm-project/pull/221794
More information about the llvm-commits
mailing list