[llvm] [llvm-strings] Add --encoding={s,S,utf8} option (PR #221794)

Harald van Dijk via llvm-commits llvm-commits at lists.llvm.org
Fri Sep 11 06:17:54 PDT 2026


https://github.com/hvdijk updated https://github.com/llvm/llvm-project/pull/221794

>From b2334d9f932e6c6f8f7e106e4678c673a19ac944 Mon Sep 17 00:00:00 2001
From: Harald van Dijk <hdijk at accesssoftek.com>
Date: Mon, 7 Sep 2026 18:33:02 +0100
Subject: [PATCH 1/5] [llvm-strings] Add --encoding & --unicode options

The --encoding and --unicode options come from GNU strings, and are
mostly compatible with it, but intentionally not fully.

The --encoding option allows specifying the following encodings:
 s   (single 7-bit byte characters)
 S   (8-bit byte characters)
 u   (UTF-8 characters)
 b/l (UTF-16 characters, big/little endian)
 B/L (UTF-32 characters, big/little endian)

--encoding=s matches GNU strings and interprets 7-bit characters as
ASCII, rejecting any bytes with the high bit set.

--encoding=S interprets characters according to the current locale's
character set. In GNU strings, its meaning depends on --unicode and
either means all characters with the high bit set are considered
printable, or means characters are interpreted as UTF-8.

--encoding=u is an extension to GNU strings, which implicitly uses UTF-8
when --unicode options are used, but does not have a dedicated encoding
value for this and puts it under --encoding=S.

--encoding=b/l/B/L match GNU strings's documentation but deviate
slightly from their implementation: they interpret characters as
UTF-16/UTF-32. Although GNU strings's manual suggests it does the same,
it actually only accepts ASCII characters stored as 16-bit/32-bit units,
whereas this implementation accepts any printable Unicode character. To
reduce the number of false positives, unlike GNU strings, this
implementation requires strings to be aligned.

--unicode=default prints characters as UTF-8. It is the default output
for all UTF encodings.

--unicode=invalid rejects any non-ASCII characters. It is the default
output for --encoding=s.

--unicode=locale prints characters in the encoding specified by the
current locale. It is the default output for --encoding=S.

--unicode=escape/highlight print non-ASCII characters as \u0000 escape
sequences, optionally highlighted if output is to a terminal.

--unicode=hex prints non-ASCII characters as their UTF-8 byte
representation in hex.

The default encoding is --encoding=S, as specified by POSIX. Although
this deviates from GNU strings, this default is meant to be a good fit
for the common case where the user's character encoding (or, in Windows
terms, the code page) matches the binary expects. --encoding=s can be
used to restore the prior behaviour.

The fact that the -n option is aliased to --bytes is confusing, but GNU
strings already uses --bytes to specify the minimum character length,
not the minimum byte length, as well.
---
 llvm/test/tools/llvm-strings/length.test      |  57 +++-
 .../tools/llvm-strings/negative-char.test     |   7 +-
 llvm/tools/llvm-strings/Opts.td               |  16 +
 llvm/tools/llvm-strings/llvm-strings.cpp      | 297 +++++++++++++++++-
 4 files changed, 357 insertions(+), 20 deletions(-)

diff --git a/llvm/test/tools/llvm-strings/length.test b/llvm/test/tools/llvm-strings/length.test
index 896917d8db90d..aaa3df198f633 100644
--- a/llvm/test/tools/llvm-strings/length.test
+++ b/llvm/test/tools/llvm-strings/length.test
@@ -1,5 +1,5 @@
 ## Show that llvm-strings prints only strings with length of at least the
-## requested number of bytes.
+## requested number of characters.
 
 RUN: echo a > %t
 RUN: echo ab >> %t
@@ -44,3 +44,58 @@ CHECK-5:      abcde
 ## Show that a non-numeric argument is rejected.
 RUN: not llvm-strings -n foo %t 2>&1 | FileCheck %s --check-prefix=ERR
 ERR: error: expected a positive integer, but got 'foo'
+
+## Show that the count is number of characters, not number of bytes.
+RUN: printf '\302\241\n' > %t
+RUN: printf '\302\241\302\242\n' >> %t
+RUN: printf '\302\241\302\242\302\243\n' >> %t
+RUN: not llvm-strings --encoding u -n 0 2>&1 %t | FileCheck --check-prefix CHECK-0 %s
+RUN: llvm-strings --encoding u -n 1 %t | FileCheck --check-prefix UTF-8-1 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding u -n 2 %t | FileCheck --check-prefix UTF-8-2 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding u -n 3 %t | FileCheck --check-prefix UTF-8-3 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding u -n 4 %t | FileCheck %s --implicit-check-not={{.}} --allow-empty
+
+RUN: printf '\241\0\n\0' > %t
+RUN: printf '\241\0\242\0\n\0' >> %t
+RUN: printf '\241\0\242\0\243\0\n\0' >> %t
+RUN: not llvm-strings --encoding l -n 0 2>&1 %t | FileCheck --check-prefix CHECK-0 %s
+RUN: llvm-strings --encoding l -n 1 %t | FileCheck --check-prefix UTF-8-1 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding l -n 2 %t | FileCheck --check-prefix UTF-8-2 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding l -n 3 %t | FileCheck --check-prefix UTF-8-3 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding l -n 4 %t | FileCheck %s --implicit-check-not={{.}} --allow-empty
+
+RUN: printf '\0\241\0\n' > %t
+RUN: printf '\0\241\0\242\0\n' >> %t
+RUN: printf '\0\241\0\242\0\243\0\n' >> %t
+RUN: not llvm-strings --encoding b -n 0 2>&1 %t | FileCheck --check-prefix CHECK-0 %s
+RUN: llvm-strings --encoding b -n 1 %t | FileCheck --check-prefix UTF-8-1 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding b -n 2 %t | FileCheck --check-prefix UTF-8-2 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding b -n 3 %t | FileCheck --check-prefix UTF-8-3 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding b -n 4 %t | FileCheck %s --implicit-check-not={{.}} --allow-empty
+
+RUN: printf '\241\0\0\0\n\0\0\0' > %t
+RUN: printf '\241\0\0\0\242\0\0\0\n\0\0\0' >> %t
+RUN: printf '\241\0\0\0\242\0\0\0\243\0\0\0\n\0\0\0' >> %t
+RUN: not llvm-strings --encoding L -n 0 2>&1 %t | FileCheck --check-prefix CHECK-0 %s
+RUN: llvm-strings --encoding L -n 1 %t | FileCheck --check-prefix UTF-8-1 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding L -n 2 %t | FileCheck --check-prefix UTF-8-2 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding L -n 3 %t | FileCheck --check-prefix UTF-8-3 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding L -n 4 %t | FileCheck %s --implicit-check-not={{.}} --allow-empty
+
+RUN: printf '\0\0\0\241\0\0\0\n' > %t
+RUN: printf '\0\0\0\241\0\0\0\242\0\0\0\n' >> %t
+RUN: printf '\0\0\0\241\0\0\0\242\0\0\0\243\0\0\0\n' >> %t
+RUN: not llvm-strings --encoding B -n 0 2>&1 %t | FileCheck --check-prefix CHECK-0 %s
+RUN: llvm-strings --encoding B -n 1 %t | FileCheck --check-prefix UTF-8-1 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding B -n 2 %t | FileCheck --check-prefix UTF-8-2 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding B -n 3 %t | FileCheck --check-prefix UTF-8-3 %s --implicit-check-not={{.}}
+RUN: llvm-strings --encoding B -n 4 %t | FileCheck %s --implicit-check-not={{.}} --allow-empty
+
+UTF-8-1:      ¡
+UTF-8-1-NEXT: ¡¢
+UTF-8-1-NEXT: ¡¢£
+
+UTF-8-2:      ¡¢
+UTF-8-2-NEXT: ¡¢£
+
+UTF-8-3:      ¡¢£
diff --git a/llvm/test/tools/llvm-strings/negative-char.test b/llvm/test/tools/llvm-strings/negative-char.test
index 5154886f9200f..3e661b0472470 100644
--- a/llvm/test/tools/llvm-strings/negative-char.test
+++ b/llvm/test/tools/llvm-strings/negative-char.test
@@ -1,6 +1,9 @@
 ## Show that llvm-strings can handle a negative signed char value (i.e. > 0x7f).
-## Such characters should form string delimiters like other unprintable ones.
+## Such characters should form string delimiters like other unprintable ones,
+## unless using an encoding where they form valid characters.
 
-# RUN: printf "z\0\200\0a\0" | llvm-strings --bytes 1 - | FileCheck %s
+# RUN: printf "z\0\200\0\302\241\0a\0" | llvm-strings --encoding=s --bytes 1 - | FileCheck %s
+# RUN: printf "z\0\200\0\302\241\0a\0" | llvm-strings --unicode=default --bytes 1 - | FileCheck %s --check-prefixes=CHECK,CHECK-UTF8
 # CHECK: z{{$}}
+# CHECK-UTF8-NEXT: {{^}}¡{{$}}
 # CHECK-NEXT: {{^}}a
diff --git a/llvm/tools/llvm-strings/Opts.td b/llvm/tools/llvm-strings/Opts.td
index 2ad77fae6c15f..cf7a406fcec1f 100644
--- a/llvm/tools/llvm-strings/Opts.td
+++ b/llvm/tools/llvm-strings/Opts.td
@@ -11,13 +11,29 @@ multiclass Eq<string name, string help> {
 
 def all : FF<"all", "Silently ignored. Present for GNU strings compatibility">;
 defm bytes : Eq<"bytes", "Print sequences of the specified length">;
+defm encoding : Eq<"encoding", [{Select the character encoding of the strings to find:
+s (7-bit characters)
+S (8-bit characters in the current locale)
+u (UTF-8)
+b (UTF-16 big endian)
+l (UTF-16 little endian)
+B (UTF-32 big endian)
+L (UTF-32 little endian)}]>;
 def help : FF<"help", "Display this help">;
 def print_file_name : Flag<["--"], "print-file-name">, HelpText<"Print the name of the file before each string">;
 defm radix : Eq<"radix", "Print the offset within the file with the specified radix: o (octal), d (decimal), x (hexadecimal)">, MetaVarName<"<radix>">;
+defm unicode : Eq<"unicode", [{Control the display of UTF-8 multibyte characters:
+default   (print UTF-8 characters)
+invalid   (reject multibyte characters)
+locale    (print characters in the current locale)
+escape    (\u0000 escape sequences)
+hex       (print UTF-8 encoding in hex)
+highlight (like escape, but highlighted)}]>;
 def version : FF<"version", "Display the version">;
 
 def : F<"a", "Alias for --all">, Alias<all>;
 def : F<"f", "Alias for --print-file-name">, Alias<print_file_name>;
 def : F<"h", "Alias for --help">, Alias<help>;
 def : JoinedOrSeparate<["-"], "n">, Alias<bytes_EQ>, HelpText<"Alias for --bytes">;
+def : JoinedOrSeparate<["-"], "e">, Alias<encoding_EQ>, HelpText<"Alias for --encoding">;
 def : JoinedOrSeparate<["-"], "t">, Alias<radix_EQ>, HelpText<"Alias for --radix">, MetaVarName<"<radix>">;
diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 9979b93de8427..655cca584437c 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -18,13 +18,17 @@
 #include "llvm/Option/ArgList.h"
 #include "llvm/Option/Option.h"
 #include "llvm/Support/CommandLine.h"
+#include "llvm/Support/ConvertUTF.h"
 #include "llvm/Support/Error.h"
 #include "llvm/Support/Format.h"
 #include "llvm/Support/InitLLVM.h"
 #include "llvm/Support/MemoryBuffer.h"
 #include "llvm/Support/Program.h"
+#include "llvm/Support/SwapByteOrder.h"
+#include "llvm/Support/Unicode.h"
 #include "llvm/Support/WithColor.h"
 #include <cctype>
+#include <locale>
 #include <string>
 
 using namespace llvm;
@@ -71,9 +75,15 @@ static cl::list<std::string> InputFileNames(cl::Positional,
 static int MinLength = 4;
 static bool PrintFileName;
 
-enum radix { none, octal, hexadecimal, decimal };
+enum class encoding { s, S, u, b, l, B, L };
+static encoding Encoding;
+
+enum class radix { none, octal, hexadecimal, decimal };
 static radix Radix;
 
+enum class unicode { default_, invalid, locale, escape, hex, highlight };
+static unicode Unicode;
+
 [[noreturn]] static void reportCmdLineError(const Twine &Message) {
   WithColor::error(errs(), ToolName) << Message << "\n";
   exit(1);
@@ -89,40 +99,224 @@ static void parseIntArg(const opt::InputArgList &Args, int ID, T &Value) {
 }
 
 static void strings(raw_ostream &OS, StringRef FileName, StringRef Contents) {
-  auto print = [&OS, FileName](unsigned Offset, StringRef L) {
-    if (L.size() < static_cast<size_t>(MinLength))
+  std::locale loc("");
+  auto &cvt = std::use_facet<std::codecvt<wchar_t, char, std::mbstate_t>>(loc);
+  auto &ctype = std::use_facet<std::ctype<wchar_t>>(loc);
+  std::mbstate_t mbs{};
+
+  const bool PrintRawBytes =
+      Encoding == encoding::s ||
+      (Encoding == encoding::S &&
+       (Unicode == unicode::invalid || Unicode == unicode::locale)) ||
+      (Encoding == encoding::u &&
+       (Unicode == unicode::default_ || Unicode == unicode::invalid));
+
+  auto read = [&cvt, &mbs](const char *&P, const char *E) -> UTF32 {
+    UTF32 Ch;
+
+    switch (Encoding) {
+    case encoding::s:
+      Ch = *P++;
+      break;
+
+    case encoding::S: {
+      const char *N;
+      wchar_t WCh;
+      wchar_t *WNext;
+      [[maybe_unused]] const auto res =
+          cvt.in(mbs, P, E, N, &WCh, &WCh + 1, WNext);
+      assert(res != std::codecvt_base::noconv);
+
+      // Only treat a non-null character as a successful conversion, as a null
+      // character may be the result of an incomplete multibyte character
+      // followed by a null byte. A null byte is safe to treat as an error, as
+      // a null byte is never printable in any locale.
+      if (WNext != &WCh && WCh) {
+        // Note: this assumes wchar_t is UCS2 or UTF32.
+        Ch = WCh;
+        P = N;
+      } else {
+        // If there was any error, skip the current byte and reset the state to
+        // allow the next byte to start a character.
+        Ch = 0;
+        mbs = {};
+        ++P;
+      }
+      break;
+    }
+
+    case encoding::u: {
+      const UTF8 *UP = reinterpret_cast<const UTF8 *>(P);
+      const UTF8 *UE = reinterpret_cast<const UTF8 *>(E);
+      UTF32 *Next = &Ch;
+      ConvertUTF8toUTF32(&UP, UE, &Next, &Ch + 1, strictConversion);
+      if (Next == &Ch || !Ch) {
+        // If there was any error, skip the current byte to allow the next to
+        // start a character.
+        Ch = 0;
+        UP = std::next(reinterpret_cast<const UTF8 *>(P));
+      }
+      P = reinterpret_cast<const char *>(UP);
+      break;
+    }
+
+    case encoding::b:
+    case encoding::l: {
+      const UTF16 *UP = reinterpret_cast<const UTF16 *>(P);
+      const UTF16 *UE = reinterpret_cast<const UTF16 *>(E);
+      const bool DoByteSwap =
+          (Encoding == encoding::b &&
+           endianness::native == endianness::little) ||
+          (Encoding == encoding::l && endianness::native == endianness::big);
+      UTF32 *Next = &Ch;
+      if (!DoByteSwap) {
+        ConvertUTF16toUTF32(&UP, UE, &Next, &Ch + 1, strictConversion);
+      } else {
+        // We never need more than two UTF16 words to make up one character.
+        const UTF16 Buf[2] = {byteswap(UP[0]),
+                              UP + 1 == UE ? UTF16(0) : byteswap(UP[1])};
+        const UTF16 *BufP = Buf;
+        ConvertUTF16toUTF32(&BufP, &Buf[2], &Next, &Ch + 1, strictConversion);
+        UP += BufP - Buf;
+      }
+      if (Next == &Ch || !Ch) {
+        // If there was any error, skip the current word to allow the next to
+        // start a character.
+        Ch = 0;
+        UP = std::next(reinterpret_cast<const UTF16 *>(P));
+      }
+      P = reinterpret_cast<const char *>(UP);
+      break;
+    }
+
+    case encoding::B:
+    case encoding::L: {
+      const UTF32 *UP = reinterpret_cast<const UTF32 *>(P);
+      const bool DoByteSwap =
+          (Encoding == encoding::B &&
+           endianness::native == endianness::little) ||
+          (Encoding == encoding::L && endianness::native == endianness::big);
+      Ch = DoByteSwap ? byteswap(*UP) : *UP;
+      ++UP;
+      P = reinterpret_cast<const char *>(UP);
+      break;
+    }
+
+    default:
+      llvm_unreachable("unhandled encoding");
+    }
+
+    return Ch;
+  };
+
+  auto print = [&OS, FileName, PrintRawBytes, &read,
+                &cvt](unsigned Offset, StringRef L, size_t N) {
+    if (N < static_cast<size_t>(MinLength))
       return;
     if (PrintFileName)
       OS << FileName << ": ";
     switch (Radix) {
-    case none:
+    case radix::none:
       break;
-    case octal:
+    case radix::octal:
       OS << format("%7o ", Offset);
       break;
-    case hexadecimal:
+    case radix::hexadecimal:
       OS << format("%7x ", Offset);
       break;
-    case decimal:
+    case radix::decimal:
       OS << format("%7u ", Offset);
       break;
     }
-    OS << L << '\n';
+
+    if (PrintRawBytes) {
+      OS << L << '\n';
+    } else {
+      mbstate_t mbs = {};
+
+      const char *P = L.begin();
+      const char *E = L.end();
+      while (P < E) {
+        const UTF32 Ch = read(P, E);
+        if (Unicode == unicode::invalid || Ch <= 0x7F) {
+          OS << (char)Ch;
+          continue;
+        }
+
+        if (Unicode == unicode::locale) {
+          // If we have a 16-bit wchar_t and the character does not fit, replace
+          // it with U+FFFD.
+          wchar_t WCh = Ch;
+          const wchar_t *WNext;
+          if ((UTF32)WCh != Ch)
+            WCh = 0xFFFD;
+          char mbstring[MB_LEN_MAX];
+          char *mbend;
+          [[maybe_unused]] const auto res =
+              cvt.out(mbs, &WCh, &WCh + 1, WNext, mbstring,
+                      &mbstring[MB_LEN_MAX], mbend);
+          assert(res == std::codecvt_base::ok);
+          OS << StringRef(mbstring, mbend - mbstring);
+          continue;
+        }
+
+        if (Unicode == unicode::escape || Unicode == unicode::highlight) {
+          WithColor COS(OS, raw_ostream::RED, false, false,
+                        Unicode == unicode::highlight ? ColorMode::Auto
+                                                      : ColorMode::Disable);
+          if (Ch <= 0xFFFF)
+            COS << "\\u" << utohexstr(Ch, false, 4);
+          else
+            COS << "\\U" << utohexstr(Ch, false, 8);
+          continue;
+        }
+
+        char UTF8[4];
+        char *UTF8end = UTF8;
+        ConvertCodePointToUTF8(Ch, UTF8end);
+        if (Unicode == unicode::hex) {
+          OS << "<0x";
+          for (unsigned char Byte : StringRef(UTF8, UTF8end - UTF8))
+            OS << utohexstr(Byte, true, 2);
+          OS << '>';
+          continue;
+        }
+
+        assert(Unicode == unicode::default_);
+        OS << StringRef(UTF8, UTF8end - UTF8);
+      }
+      OS << '\n';
+    }
   };
 
+  std::size_t NumPrintable = 0;
+
   const char *B = Contents.begin();
-  const char *P = nullptr, *E = nullptr, *S = nullptr;
-  for (P = Contents.begin(), E = Contents.end(); P < E; ++P) {
-    if (isPrint(*P) || *P == '\t') {
+  const char *P = Contents.begin(), *E = Contents.end(), *S = nullptr;
+  for (P = Contents.begin(), E = Contents.end(); P < E;) {
+    const char *N = P;
+    UTF32 Ch = read(N, E);
+
+    const bool Printable =
+        Ch == '\t' ||
+        (Unicode == unicode::invalid ? Ch <= 0x7f && isPrint(Ch)
+         : Encoding == encoding::S   ? ctype.is(std::ctype_base::print, Ch)
+                                     : sys::unicode::isPrintable(Ch));
+
+    if (Printable) {
       if (S == nullptr)
         S = P;
+      ++NumPrintable;
     } else if (S) {
-      print(S - B, StringRef(S, P - S));
+      print(S - B, StringRef(S, P - S), NumPrintable);
       S = nullptr;
+      NumPrintable = 0;
     }
+
+    P = N;
   }
   if (S)
-    print(S - B, StringRef(S, E - S));
+    print(S - B, StringRef(S, E - S), NumPrintable);
 }
 
 int main(int argc, char **argv) {
@@ -149,20 +343,89 @@ int main(int argc, char **argv) {
     return 0;
   }
 
+  const auto *EncodingArg = Args.getLastArg(OPT_encoding_EQ);
+  const auto *UnicodeArg = Args.getLastArg(OPT_unicode_EQ);
+
+  Encoding =
+      EncodingArg
+          ? llvm::StringSwitch<encoding>(EncodingArg->getValue())
+                .Case("s", encoding::s)
+                .Case("S", encoding::S)
+                .Case("u", encoding::u)
+                .Case("b", encoding::b)
+                .Case("l", encoding::l)
+                .Case("B", encoding::B)
+                .Case("L", encoding::L)
+                .Predicate(
+                    [](StringRef) -> bool {
+                      reportCmdLineError(
+                          "--encoding value should be one of: "
+                          "'s' (7-bit characters), "
+                          "'S' (8-bit characters), "
+                          "'u' (UTF-8 characters), "
+                          "'b', 'l' (16-bit big/little endian characters), "
+                          "'B', 'L' (32-bit big/little endian characters)");
+                    },
+                    encoding{})
+                .DefaultUnreachable()
+          : encoding::S;
   parseIntArg(Args, OPT_bytes_EQ, MinLength);
   PrintFileName = Args.hasArg(OPT_print_file_name);
   StringRef R = Args.getLastArgValue(OPT_radix_EQ);
   if (R.empty())
-    Radix = none;
+    Radix = radix::none;
   else if (R == "o")
-    Radix = octal;
+    Radix = radix::octal;
   else if (R == "d")
-    Radix = decimal;
+    Radix = radix::decimal;
   else if (R == "x")
-    Radix = hexadecimal;
+    Radix = radix::hexadecimal;
   else
     reportCmdLineError("--radix value should be one of: '' (no offset), 'o' "
                        "(octal), 'd' (decimal), 'x' (hexadecimal)");
+  Unicode = UnicodeArg ? llvm::StringSwitch<unicode>(UnicodeArg->getValue())
+                             .Case("default", unicode::default_)
+                             .Case("invalid", unicode::invalid)
+                             .Case("locale", unicode::locale)
+                             .Case("escape", unicode::escape)
+                             .Case("hex", unicode::hex)
+                             .Case("highlight", unicode::highlight)
+                             .Predicate(
+                                 [](StringRef) -> bool {
+                                   reportCmdLineError(
+                                       "--unicode value should be one of: "
+                                       "default, "
+                                       "invalid, "
+                                       "locale, "
+                                       "escape, "
+                                       "hex, "
+                                       "highlight");
+                                 },
+                                 unicode::default_)
+                             .DefaultUnreachable()
+                       : unicode::locale;
+
+  // The defaults are ugly to maintain compatibility with GNU strings in the
+  // common cases. The general idea is that --encoding specifies the input
+  // encoding, --unicode specifies the output encoding, but when one is
+  // specified, the other defaults to the best match with the limitation that
+  // the output encoding is never UTF-16 or UTF-32.
+  if (!EncodingArg && Unicode != unicode::invalid && Unicode != unicode::locale)
+    Encoding = encoding::u;
+
+  if (!UnicodeArg && Encoding != encoding::S)
+    Unicode = unicode::default_;
+
+  // With --encoding=s, all output encodings would give the same results. Use
+  // the simplest one.
+  if (Encoding == encoding::s)
+    Unicode = unicode::invalid;
+
+  // With --encoding=u --unicode=invalid, we do not need to spend time decoding
+  // UTF-8 only to reject whatever results we get, we can reject multibyte
+  // characters right away.
+  if (Encoding == encoding::u && Unicode == unicode::invalid)
+    Encoding = encoding::s;
 
   if (MinLength == 0) {
     errs() << "invalid minimum string length 0\n";

>From 11d4e557a1824d7cd71f720cbed1e9a162e69f50 Mon Sep 17 00:00:00 2001
From: Harald van Dijk <hdijk at accesssoftek.com>
Date: Mon, 7 Sep 2026 19:46:55 +0100
Subject: [PATCH 2/5] Remove llvm_unreachable to suppress compiler warning

---
 llvm/tools/llvm-strings/llvm-strings.cpp | 3 ---
 1 file changed, 3 deletions(-)

diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 655cca584437c..79baa08cc6801 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -201,9 +201,6 @@ static void strings(raw_ostream &OS, StringRef FileName, StringRef Contents) {
       P = reinterpret_cast<const char *>(UP);
       break;
     }
-
-    default:
-      llvm_unreachable("unhandled encoding");
     }
 
     return Ch;

>From d6cb511d4d1f18486f1221f7b57ea0cca7d7d026 Mon Sep 17 00:00:00 2001
From: Harald van Dijk <hdijk at accesssoftek.com>
Date: Tue, 8 Sep 2026 14:47:58 +0100
Subject: [PATCH 3/5] Update manpage, remove --encoding=b/l/B/L and --unicode,
 add test

---
 llvm/docs/CommandGuide/llvm-strings.md        |  15 +-
 llvm/test/tools/llvm-strings/encoding.test    |  28 +++
 llvm/test/tools/llvm-strings/length.test      |  36 ---
 .../tools/llvm-strings/negative-char.test     |   2 +-
 llvm/tools/llvm-strings/Opts.td               |  17 +-
 llvm/tools/llvm-strings/llvm-strings.cpp      | 208 ++----------------
 6 files changed, 67 insertions(+), 239 deletions(-)
 create mode 100644 llvm/test/tools/llvm-strings/encoding.test

diff --git a/llvm/docs/CommandGuide/llvm-strings.md b/llvm/docs/CommandGuide/llvm-strings.md
index da9d0a03d9fd5..6789a316a6684 100644
--- a/llvm/docs/CommandGuide/llvm-strings.md
+++ b/llvm/docs/CommandGuide/llvm-strings.md
@@ -12,8 +12,8 @@
 {program}`llvm-strings` is a tool intended as a drop-in replacement for GNU's
 {program}`strings`, which looks for printable strings in files and writes them
 to the standard output stream. A printable string is any sequence of four (by
-default) or more printable ASCII characters. The end of the file, or any other
-byte, terminates the current sequence.
+default) or more printable characters. The end of the file, or any other byte,
+terminates the current sequence.
 
 {program}`llvm-strings` looks for strings in each `input` file specified.
 Unlike GNU {program}`strings` it looks in the entire input file, regardless of
@@ -40,8 +40,15 @@ Silently ignored. Present for GNU {program}`strings` compatibility.
 :::
 
 :::{option} --bytes=<length>, -n
-Set the minimum number of printable ASCII characters required for a sequence of
-bytes to be considered a string. The default value is 4.
+Set the minimum number of printable characters required for a sequence to be
+considered a string. The default value is 4.
+:::
+
+:::{option} --encoding=<encoding>, -e
+Specifies the encoding of the input file. Valid arguments are:
+- s (ASCII characters)
+- S (characters in the system's or user's selected character set)
+- u (UTF-8 characters)
 :::
 
 :::{option} --help, -h
diff --git a/llvm/test/tools/llvm-strings/encoding.test b/llvm/test/tools/llvm-strings/encoding.test
new file mode 100644
index 0000000000000..7355516e693cf
--- /dev/null
+++ b/llvm/test/tools/llvm-strings/encoding.test
@@ -0,0 +1,28 @@
+## Show that llvm-strings uses the specified encoding.
+
+RUN: echo a > %t
+RUN: echo ab >> %t
+RUN: echo abc >> %t
+RUN: echo abcd >> %t
+RUN: echo abcd€ >> %t
+
+# Check that long form options work.
+RUN: llvm-strings --encoding s 2>&1 %t | FileCheck --check-prefixes CHECK,CHECK-ASCII %s
+RUN: llvm-strings --encoding S 2>&1 %t | FileCheck                                    %s
+RUN: llvm-strings --encoding u 2>&1 %t | FileCheck --check-prefixes CHECK,CHECK-UTF8  %s
+RUN: not llvm-strings --encoding x 2>&1 %t | FileCheck --check-prefixes CHECK-ERROR       %s
+
+# Check that short form options work.
+RUN: llvm-strings -e s 2>&1 %t | FileCheck --check-prefixes CHECK,CHECK-ASCII %s
+RUN: llvm-strings -e S 2>&1 %t | FileCheck                                    %s
+RUN: llvm-strings -e u 2>&1 %t | FileCheck --check-prefixes CHECK,CHECK-UTF8  %s
+
+CHECK: abcd
+CHECK-ASCII-NOT: abcd€
+CHECK-UTF8: abcd€
+
+# Check that invalid encodings are rejected.
+RUN: not llvm-strings --encoding x 2>&1 %t | FileCheck --check-prefix CHECK-ERROR %s
+RUN: not llvm-strings -e x 2>&1 %t | FileCheck --check-prefix CHECK-ERROR %s
+
+CHECK-ERROR: error: --encoding value
diff --git a/llvm/test/tools/llvm-strings/length.test b/llvm/test/tools/llvm-strings/length.test
index aaa3df198f633..5ec95baea317e 100644
--- a/llvm/test/tools/llvm-strings/length.test
+++ b/llvm/test/tools/llvm-strings/length.test
@@ -55,42 +55,6 @@ RUN: llvm-strings --encoding u -n 2 %t | FileCheck --check-prefix UTF-8-2 %s --i
 RUN: llvm-strings --encoding u -n 3 %t | FileCheck --check-prefix UTF-8-3 %s --implicit-check-not={{.}}
 RUN: llvm-strings --encoding u -n 4 %t | FileCheck %s --implicit-check-not={{.}} --allow-empty
 
-RUN: printf '\241\0\n\0' > %t
-RUN: printf '\241\0\242\0\n\0' >> %t
-RUN: printf '\241\0\242\0\243\0\n\0' >> %t
-RUN: not llvm-strings --encoding l -n 0 2>&1 %t | FileCheck --check-prefix CHECK-0 %s
-RUN: llvm-strings --encoding l -n 1 %t | FileCheck --check-prefix UTF-8-1 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding l -n 2 %t | FileCheck --check-prefix UTF-8-2 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding l -n 3 %t | FileCheck --check-prefix UTF-8-3 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding l -n 4 %t | FileCheck %s --implicit-check-not={{.}} --allow-empty
-
-RUN: printf '\0\241\0\n' > %t
-RUN: printf '\0\241\0\242\0\n' >> %t
-RUN: printf '\0\241\0\242\0\243\0\n' >> %t
-RUN: not llvm-strings --encoding b -n 0 2>&1 %t | FileCheck --check-prefix CHECK-0 %s
-RUN: llvm-strings --encoding b -n 1 %t | FileCheck --check-prefix UTF-8-1 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding b -n 2 %t | FileCheck --check-prefix UTF-8-2 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding b -n 3 %t | FileCheck --check-prefix UTF-8-3 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding b -n 4 %t | FileCheck %s --implicit-check-not={{.}} --allow-empty
-
-RUN: printf '\241\0\0\0\n\0\0\0' > %t
-RUN: printf '\241\0\0\0\242\0\0\0\n\0\0\0' >> %t
-RUN: printf '\241\0\0\0\242\0\0\0\243\0\0\0\n\0\0\0' >> %t
-RUN: not llvm-strings --encoding L -n 0 2>&1 %t | FileCheck --check-prefix CHECK-0 %s
-RUN: llvm-strings --encoding L -n 1 %t | FileCheck --check-prefix UTF-8-1 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding L -n 2 %t | FileCheck --check-prefix UTF-8-2 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding L -n 3 %t | FileCheck --check-prefix UTF-8-3 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding L -n 4 %t | FileCheck %s --implicit-check-not={{.}} --allow-empty
-
-RUN: printf '\0\0\0\241\0\0\0\n' > %t
-RUN: printf '\0\0\0\241\0\0\0\242\0\0\0\n' >> %t
-RUN: printf '\0\0\0\241\0\0\0\242\0\0\0\243\0\0\0\n' >> %t
-RUN: not llvm-strings --encoding B -n 0 2>&1 %t | FileCheck --check-prefix CHECK-0 %s
-RUN: llvm-strings --encoding B -n 1 %t | FileCheck --check-prefix UTF-8-1 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding B -n 2 %t | FileCheck --check-prefix UTF-8-2 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding B -n 3 %t | FileCheck --check-prefix UTF-8-3 %s --implicit-check-not={{.}}
-RUN: llvm-strings --encoding B -n 4 %t | FileCheck %s --implicit-check-not={{.}} --allow-empty
-
 UTF-8-1:      ¡
 UTF-8-1-NEXT: ¡¢
 UTF-8-1-NEXT: ¡¢£
diff --git a/llvm/test/tools/llvm-strings/negative-char.test b/llvm/test/tools/llvm-strings/negative-char.test
index 3e661b0472470..8e61700e07e8a 100644
--- a/llvm/test/tools/llvm-strings/negative-char.test
+++ b/llvm/test/tools/llvm-strings/negative-char.test
@@ -3,7 +3,7 @@
 ## unless using an encoding where they form valid characters.
 
 # RUN: printf "z\0\200\0\302\241\0a\0" | llvm-strings --encoding=s --bytes 1 - | FileCheck %s
-# RUN: printf "z\0\200\0\302\241\0a\0" | llvm-strings --unicode=default --bytes 1 - | FileCheck %s --check-prefixes=CHECK,CHECK-UTF8
+# RUN: printf "z\0\200\0\302\241\0a\0" | llvm-strings --encoding=u --bytes 1 - | FileCheck %s --check-prefixes=CHECK,CHECK-UTF8
 # CHECK: z{{$}}
 # CHECK-UTF8-NEXT: {{^}}¡{{$}}
 # CHECK-NEXT: {{^}}a
diff --git a/llvm/tools/llvm-strings/Opts.td b/llvm/tools/llvm-strings/Opts.td
index cf7a406fcec1f..dd4f4349d7eed 100644
--- a/llvm/tools/llvm-strings/Opts.td
+++ b/llvm/tools/llvm-strings/Opts.td
@@ -12,23 +12,12 @@ multiclass Eq<string name, string help> {
 def all : FF<"all", "Silently ignored. Present for GNU strings compatibility">;
 defm bytes : Eq<"bytes", "Print sequences of the specified length">;
 defm encoding : Eq<"encoding", [{Select the character encoding of the strings to find:
-s (7-bit characters)
-S (8-bit characters in the current locale)
-u (UTF-8)
-b (UTF-16 big endian)
-l (UTF-16 little endian)
-B (UTF-32 big endian)
-L (UTF-32 little endian)}]>;
+s (ASCII characters)
+S (characters in the system's or user's character set)
+u (UTF-8 characters)}]>;
 def help : FF<"help", "Display this help">;
 def print_file_name : Flag<["--"], "print-file-name">, HelpText<"Print the name of the file before each string">;
 defm radix : Eq<"radix", "Print the offset within the file with the specified radix: o (octal), d (decimal), x (hexadecimal)">, MetaVarName<"<radix>">;
-defm unicode : Eq<"unicode", [{Control the display of UTF-8 multibyte characters:
-default   (print UTF-8 characters)
-invalid   (reject multibyte characters)
-locale    (print characters in the current locale)
-escape    (\u0000 escape sequences)
-hex       (print UTF-8 encoding in hex)
-highlight (like escape, but highlighted)}]>;
 def version : FF<"version", "Display the version">;
 
 def : F<"a", "Alias for --all">, Alias<all>;
diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index 79baa08cc6801..bb61425f790d3 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -75,15 +75,12 @@ static cl::list<std::string> InputFileNames(cl::Positional,
 static int MinLength = 4;
 static bool PrintFileName;
 
-enum class encoding { s, S, u, b, l, B, L };
+enum class encoding { s, S, u };
 static encoding Encoding;
 
 enum class radix { none, octal, hexadecimal, decimal };
 static radix Radix;
 
-enum class unicode { default_, invalid, locale, escape, hex, highlight };
-static unicode Unicode;
-
 [[noreturn]] static void reportCmdLineError(const Twine &Message) {
   WithColor::error(errs(), ToolName) << Message << "\n";
   exit(1);
@@ -104,13 +101,6 @@ static void strings(raw_ostream &OS, StringRef FileName, StringRef Contents) {
   auto &ctype = std::use_facet<std::ctype<wchar_t>>(loc);
   std::mbstate_t mbs{};
 
-  const bool PrintRawBytes =
-      Encoding == encoding::s ||
-      (Encoding == encoding::S &&
-       (Unicode == unicode::invalid || Unicode == unicode::locale)) ||
-      (Encoding == encoding::u &&
-       (Unicode == unicode::default_ || Unicode == unicode::invalid));
-
   auto read = [&cvt, &mbs](const char *&P, const char *E) -> UTF32 {
     UTF32 Ch;
 
@@ -159,55 +149,13 @@ static void strings(raw_ostream &OS, StringRef FileName, StringRef Contents) {
       P = reinterpret_cast<const char *>(UP);
       break;
     }
-
-    case encoding::b:
-    case encoding::l: {
-      const UTF16 *UP = reinterpret_cast<const UTF16 *>(P);
-      const UTF16 *UE = reinterpret_cast<const UTF16 *>(E);
-      const bool DoByteSwap =
-          (Encoding == encoding::b &&
-           endianness::native == endianness::little) ||
-          (Encoding == encoding::l && endianness::native == endianness::big);
-      UTF32 *Next = &Ch;
-      if (!DoByteSwap) {
-        ConvertUTF16toUTF32(&UP, UE, &Next, &Ch + 1, strictConversion);
-      } else {
-        // We never need more than two UTF16 words to make up one character.
-        const UTF16 Buf[2] = {byteswap(UP[0]),
-                              UP + 1 == UE ? UTF16(0) : byteswap(UP[1])};
-        const UTF16 *BufP = Buf;
-        ConvertUTF16toUTF32(&BufP, &Buf[2], &Next, &Ch + 1, strictConversion);
-        UP += BufP - Buf;
-      }
-      if (Next == &Ch || !Ch) {
-        // If there was any error, skip the current word to allow the next to
-        // start a character.
-        Ch = 0;
-        UP = std::next(reinterpret_cast<const UTF16 *>(P));
-      }
-      P = reinterpret_cast<const char *>(UP);
-      break;
-    }
-
-    case encoding::B:
-    case encoding::L: {
-      const UTF32 *UP = reinterpret_cast<const UTF32 *>(P);
-      const bool DoByteSwap =
-          (Encoding == encoding::B &&
-           endianness::native == endianness::little) ||
-          (Encoding == encoding::L && endianness::native == endianness::big);
-      Ch = DoByteSwap ? byteswap(*UP) : *UP;
-      ++UP;
-      P = reinterpret_cast<const char *>(UP);
-      break;
-    }
     }
 
     return Ch;
   };
 
-  auto print = [&OS, FileName, PrintRawBytes, &read,
-                &cvt](unsigned Offset, StringRef L, size_t N) {
+  auto print = [&OS, FileName, &read, &cvt](unsigned Offset, StringRef L,
+                                            size_t N) {
     if (N < static_cast<size_t>(MinLength))
       return;
     if (PrintFileName)
@@ -226,64 +174,7 @@ static void strings(raw_ostream &OS, StringRef FileName, StringRef Contents) {
       break;
     }
 
-    if (PrintRawBytes) {
-      OS << L << '\n';
-    } else {
-      mbstate_t mbs = {};
-
-      const char *P = L.begin();
-      const char *E = L.end();
-      while (P < E) {
-        const UTF32 Ch = read(P, E);
-        if (Unicode == unicode::invalid || Ch <= 0x7F) {
-          OS << (char)Ch;
-          continue;
-        }
-
-        if (Unicode == unicode::locale) {
-          // If we have a 16-bit wchar_t and the character does not fit, replace
-          // it with U+FFFD.
-          wchar_t WCh = Ch;
-          const wchar_t *WNext;
-          if ((UTF32)WCh != Ch)
-            WCh = 0xFFFD;
-          char mbstring[MB_LEN_MAX];
-          char *mbend;
-          [[maybe_unused]] const auto res =
-              cvt.out(mbs, &WCh, &WCh + 1, WNext, mbstring,
-                      &mbstring[MB_LEN_MAX], mbend);
-          assert(res == std::codecvt_base::ok);
-          OS << StringRef(mbstring, mbend - mbstring);
-          continue;
-        }
-
-        if (Unicode == unicode::escape || Unicode == unicode::highlight) {
-          WithColor COS(OS, raw_ostream::RED, false, false,
-                        Unicode == unicode::highlight ? ColorMode::Auto
-                                                      : ColorMode::Disable);
-          if (Ch <= 0xFFFF)
-            COS << "\\u" << utohexstr(Ch, false, 4);
-          else
-            COS << "\\U" << utohexstr(Ch, false, 8);
-          continue;
-        }
-
-        char UTF8[4];
-        char *UTF8end = UTF8;
-        ConvertCodePointToUTF8(Ch, UTF8end);
-        if (Unicode == unicode::hex) {
-          OS << "<0x";
-          for (unsigned char Byte : StringRef(UTF8, UTF8end - UTF8))
-            OS << utohexstr(Byte, true, 2);
-          OS << '>';
-          continue;
-        }
-
-        assert(Unicode == unicode::default_);
-        OS << StringRef(UTF8, UTF8end - UTF8);
-      }
-      OS << '\n';
-    }
+    OS << L << '\n';
   };
 
   std::size_t NumPrintable = 0;
@@ -296,9 +187,9 @@ static void strings(raw_ostream &OS, StringRef FileName, StringRef Contents) {
 
     const bool Printable =
         Ch == '\t' ||
-        (Unicode == unicode::invalid ? Ch <= 0x7f && isPrint(Ch)
-         : Encoding == encoding::S   ? ctype.is(std::ctype_base::print, Ch)
-                                     : sys::unicode::isPrintable(Ch));
+        (Encoding == encoding::s   ? Ch <= 0x7f && isPrint(Ch)
+         : Encoding == encoding::S ? ctype.is(std::ctype_base::print, Ch)
+                                   : sys::unicode::isPrintable(Ch));
 
     if (Printable) {
       if (S == nullptr)
@@ -341,31 +232,23 @@ int main(int argc, char **argv) {
   }
 
   const auto *EncodingArg = Args.getLastArg(OPT_encoding_EQ);
-  const auto *UnicodeArg = Args.getLastArg(OPT_unicode_EQ);
-
-  Encoding =
-      EncodingArg
-          ? llvm::StringSwitch<encoding>(EncodingArg->getValue())
-                .Case("s", encoding::s)
-                .Case("S", encoding::S)
-                .Case("u", encoding::u)
-                .Case("b", encoding::b)
-                .Case("l", encoding::l)
-                .Case("B", encoding::B)
-                .Case("L", encoding::L)
-                .Predicate(
-                    [](StringRef) -> bool {
-                      reportCmdLineError(
-                          "--encoding value should be one of: "
-                          "'s' (7-bit characters), "
-                          "'S' (8-bit characters), "
-                          "'u' (UTF-8 characters), "
-                          "'b', 'l' (16-bit big/little endian characters), "
-                          "'B', 'L' (32-bit big/little endian characters)");
-                    },
-                    encoding{})
-                .DefaultUnreachable()
-          : encoding::S;
+
+  Encoding = EncodingArg ? llvm::StringSwitch<encoding>(EncodingArg->getValue())
+                               .Case("s", encoding::s)
+                               .Case("S", encoding::S)
+                               .Case("u", encoding::u)
+                               .Predicate(
+                                   [](StringRef) -> bool {
+                                     reportCmdLineError(
+                                         "--encoding value should be one of: "
+                                         "'s' (ASCII characters), "
+                                         "'S' (characters in the system's or "
+                                         "user's selected character set), "
+                                         "'u' (UTF-8 characters)");
+                                   },
+                                   encoding{})
+                               .DefaultUnreachable()
+                         : encoding::S;
   parseIntArg(Args, OPT_bytes_EQ, MinLength);
   PrintFileName = Args.hasArg(OPT_print_file_name);
   StringRef R = Args.getLastArgValue(OPT_radix_EQ);
@@ -380,49 +263,6 @@ int main(int argc, char **argv) {
   else
     reportCmdLineError("--radix value should be one of: '' (no offset), 'o' "
                        "(octal), 'd' (decimal), 'x' (hexadecimal)");
-  Unicode = UnicodeArg ? llvm::StringSwitch<unicode>(UnicodeArg->getValue())
-                             .Case("default", unicode::default_)
-                             .Case("invalid", unicode::invalid)
-                             .Case("locale", unicode::locale)
-                             .Case("escape", unicode::escape)
-                             .Case("hex", unicode::hex)
-                             .Case("highlight", unicode::highlight)
-                             .Predicate(
-                                 [](StringRef) -> bool {
-                                   reportCmdLineError(
-                                       "--unicode value should be one of: "
-                                       "default, "
-                                       "invalid, "
-                                       "locale, "
-                                       "escape, "
-                                       "hex, "
-                                       "highlight");
-                                 },
-                                 unicode::default_)
-                             .DefaultUnreachable()
-                       : unicode::locale;
-
-  // The defaults are ugly to maintain compatibility with GNU strings in the
-  // common cases. The general idea is that --encoding specifies the input
-  // encoding, --unicode specifies the output encoding, but when one is
-  // specified, the other defaults to the best match with the limitation that
-  // the output encoding is never UTF-16 or UTF-32.
-  if (!EncodingArg && Unicode != unicode::invalid && Unicode != unicode::locale)
-    Encoding = encoding::u;
-
-  if (!UnicodeArg && Encoding != encoding::S)
-    Unicode = unicode::default_;
-
-  // With --encoding=s, all output encodings would give the same results. Use
-  // the simplest one.
-  if (Encoding == encoding::s)
-    Unicode = unicode::invalid;
-
-  // With --encoding=u --unicode=invalid, we do not need to spend time decoding
-  // UTF-8 only to reject whatever results we get, we can reject multibyte
-  // characters right away.
-  if (Encoding == encoding::u && Unicode == unicode::invalid)
-    Encoding = encoding::s;
 
   if (MinLength == 0) {
     errs() << "invalid minimum string length 0\n";

>From 16213a08dc2fb154d0d3c105b611093f194c84ba Mon Sep 17 00:00:00 2001
From: Harald van Dijk <hdijk at accesssoftek.com>
Date: Tue, 8 Sep 2026 16:04:04 +0100
Subject: [PATCH 4/5] Remove unused captures

---
 llvm/tools/llvm-strings/llvm-strings.cpp | 3 +--
 1 file changed, 1 insertion(+), 2 deletions(-)

diff --git a/llvm/tools/llvm-strings/llvm-strings.cpp b/llvm/tools/llvm-strings/llvm-strings.cpp
index bb61425f790d3..2444847ef812c 100644
--- a/llvm/tools/llvm-strings/llvm-strings.cpp
+++ b/llvm/tools/llvm-strings/llvm-strings.cpp
@@ -154,8 +154,7 @@ static void strings(raw_ostream &OS, StringRef FileName, StringRef Contents) {
     return Ch;
   };
 
-  auto print = [&OS, FileName, &read, &cvt](unsigned Offset, StringRef L,
-                                            size_t N) {
+  auto print = [&OS, FileName](unsigned Offset, StringRef L, size_t N) {
     if (N < static_cast<size_t>(MinLength))
       return;
     if (PrintFileName)

>From 02654008d5f94523fa2fc8e8ac521af576b3653d Mon Sep 17 00:00:00 2001
From: Harald van Dijk <hdijk at accesssoftek.com>
Date: Fri, 11 Sep 2026 14:17:34 +0100
Subject: [PATCH 5/5] Apply review comments to docs, encoding.test

---
 llvm/docs/CommandGuide/llvm-strings.md     |  6 +++---
 llvm/test/tools/llvm-strings/encoding.test | 17 +++++++++++------
 2 files changed, 14 insertions(+), 9 deletions(-)

diff --git a/llvm/docs/CommandGuide/llvm-strings.md b/llvm/docs/CommandGuide/llvm-strings.md
index e72cea195d6e5..5a45dcd426c05 100644
--- a/llvm/docs/CommandGuide/llvm-strings.md
+++ b/llvm/docs/CommandGuide/llvm-strings.md
@@ -12,8 +12,8 @@
 {program}`llvm-strings` is a tool intended as a drop-in replacement for GNU's
 {program}`strings`, which looks for printable strings in files and writes them
 to the standard output stream. A printable string is any sequence of four (by
-default) or more printable characters. The end of the file, or any other byte,
-terminates the current sequence.
+default) or more printable characters. The end of the file, or any unprintable
+byte sequence, terminates the current sequence.
 
 {program}`llvm-strings` looks for strings in each `input` file specified.
 Unlike GNU {program}`strings` it looks in the entire input file, regardless of
@@ -51,7 +51,7 @@ characters.
 :::{option} --encoding=<encoding>, -e
 Specifies the encoding of the input file. Valid arguments are:
 - `s` (ASCII)
-- `S` (default character set)
+- `S` (current locale's character set)
 - `utf8` (UTF-8)
 :::
 
diff --git a/llvm/test/tools/llvm-strings/encoding.test b/llvm/test/tools/llvm-strings/encoding.test
index 46a58789c5bb2..b07df9209d585 100644
--- a/llvm/test/tools/llvm-strings/encoding.test
+++ b/llvm/test/tools/llvm-strings/encoding.test
@@ -1,31 +1,36 @@
-## Show that llvm-strings uses the specified encoding.
+# Show that llvm-strings uses the specified encoding.
 RUN: echo a > %t
 RUN: echo ab >> %t
 RUN: echo abc >> %t
 RUN: echo abcd >> %t
 RUN: echo abcd€ >> %t
 
-## Check that long form options work.
+# Check that long form options work.
 RUN: llvm-strings --encoding s %t | FileCheck --check-prefixes CHECK,CHECK-ASCII --strict-whitespace --match-full-lines %s
 RUN: llvm-strings --encoding S %t | FileCheck --check-prefixes CHECK,CHECK-LOCALE --strict-whitespace --match-full-lines %s
 RUN: llvm-strings --encoding utf8 %t | FileCheck --check-prefixes CHECK,CHECK-UTF8 --strict-whitespace --match-full-lines %s
 
-## Check that short form options work.
+# Check that short form options work.
 RUN: llvm-strings -e s %t | FileCheck --check-prefixes CHECK,CHECK-ASCII --strict-whitespace --match-full-lines %s
 RUN: llvm-strings -e S %t | FileCheck --check-prefixes CHECK,CHECK-LOCALE --strict-whitespace --match-full-lines %s
 RUN: llvm-strings -e utf8 %t | FileCheck --check-prefixes CHECK,CHECK-UTF8 --strict-whitespace --match-full-lines %s
 
-## Check that the default output matches that of -e S.
+# Check that the default output matches that of -e S.
 RUN: llvm-strings -e S %t > %t.1
 RUN: llvm-strings %t > %t.2
 RUN: cmp %t.1 %t.2
 
+# The first line containing just abcd should always be matched.
 CHECK:abcd
-CHECK-ASCII-NOT:abcd€
+# In -e s, the bytes of € should not be included.
+CHECK-ASCII:abcd
+# In -e S, any bytes of € may or may not be included depending on the locale
+# in effect. The locale in effect may also be altered by llvm-lit.
 CHECK-LOCALE:abcd{{.*}}
+# In -e utf8, all bytes of € should be included.
 CHECK-UTF8:abcd€
 
-## Check that invalid --encoding values are rejected.
+# Check that invalid --encoding values are rejected.
 RUN: not llvm-strings --encoding x 2>&1 %t | FileCheck --check-prefix CHECK-ERROR %s
 RUN: not llvm-strings -e x 2>&1 %t | FileCheck --check-prefix CHECK-ERROR %s
 



More information about the llvm-commits mailing list