[llvm] [MCParser] Deduplicate isIdentifierChar. (#218100) (PR #218851)

via llvm-commits llvm-commits at lists.llvm.org
Wed Aug 26 00:19:20 PDT 2026


https://github.com/zhangweize9-cyber updated https://github.com/llvm/llvm-project/pull/218851

>From d1126e138830dfa96cf88abb7c8e6fb7f8b00e5c Mon Sep 17 00:00:00 2001
From: zhangweize9-cyber <zhangweize9 at gmail.com>
Date: Sat, 22 Aug 2026 13:06:39 +0800
Subject: [PATCH] [MCParser] Deduplicate isIdentifierChar. (#218100)

---
 llvm/include/llvm/MC/MCParser/AsmLexer.h |  8 ++++++++
 llvm/lib/MC/MCParser/AsmLexer.cpp        |  6 ------
 llvm/lib/MC/MCParser/AsmParser.cpp       | 18 +++++-------------
 3 files changed, 13 insertions(+), 19 deletions(-)

diff --git a/llvm/include/llvm/MC/MCParser/AsmLexer.h b/llvm/include/llvm/MC/MCParser/AsmLexer.h
index 7848fc706d5eb..2cd9a0f439cfb 100644
--- a/llvm/include/llvm/MC/MCParser/AsmLexer.h
+++ b/llvm/include/llvm/MC/MCParser/AsmLexer.h
@@ -15,6 +15,7 @@
 
 #include "llvm/ADT/ArrayRef.h"
 #include "llvm/ADT/SmallVector.h"
+#include "llvm/ADT/StringExtras.h"
 #include "llvm/ADT/StringRef.h"
 #include "llvm/MC/MCAsmMacro.h"
 #include "llvm/Support/Compiler.h"
@@ -130,6 +131,13 @@ class AsmLexer {
     return Tok;
   }
 
+  /// LexIdentifier: [a-zA-Z_$.@?][a-zA-Z0-9_$.@#?]*
+  static LLVM_ATTRIBUTE_ALWAYS_INLINE bool
+  isIdentifierChar(char C, bool AllowAt = false, bool AllowHash = false) {
+    return llvm::isAlnum(C) || C == '_' || C == '$' || C == '.' || C == '?' ||
+           (AllowAt && C == '@') || (AllowHash && C == '#');
+  }
+
   /// Look ahead an arbitrary number of tokens.
   LLVM_ABI size_t peekTokens(MutableArrayRef<AsmToken> Buf,
                              bool ShouldSkipSpace = true);
diff --git a/llvm/lib/MC/MCParser/AsmLexer.cpp b/llvm/lib/MC/MCParser/AsmLexer.cpp
index 8e4b7be98bdb6..cb7bb903276d4 100644
--- a/llvm/lib/MC/MCParser/AsmLexer.cpp
+++ b/llvm/lib/MC/MCParser/AsmLexer.cpp
@@ -228,12 +228,6 @@ AsmToken AsmLexer::LexHexFloatLiteral(bool NoIntDigits) {
   return AsmToken(AsmToken::Real, StringRef(TokStart, CurPtr - TokStart));
 }
 
-/// LexIdentifier: [a-zA-Z_$.@?][a-zA-Z0-9_$.@#?]*
-static bool isIdentifierChar(char C, bool AllowAt, bool AllowHash) {
-  return isAlnum(C) || C == '_' || C == '$' || C == '.' || C == '?' ||
-         (AllowAt && C == '@') || (AllowHash && C == '#');
-}
-
 AsmToken AsmLexer::LexIdentifier() {
   // Check for floating point literals.
   if (CurPtr[-1] == '.' && isDigit(*CurPtr)) {
diff --git a/llvm/lib/MC/MCParser/AsmParser.cpp b/llvm/lib/MC/MCParser/AsmParser.cpp
index 320a3d896e9d0..610bfdbe1de89 100644
--- a/llvm/lib/MC/MCParser/AsmParser.cpp
+++ b/llvm/lib/MC/MCParser/AsmParser.cpp
@@ -2459,15 +2459,6 @@ void AsmParser::DiagHandler(const SMDiagnostic &Diag, void *Context) {
     Parser->getContext().diagnose(NewDiag);
 }
 
-// FIXME: This is mostly duplicated from the function in AsmLexer.cpp. The
-// difference being that that function accepts '@' as part of identifiers and
-// we can't do that. AsmLexer.cpp should probably be changed to handle
-// '@' as a special case when needed.
-static bool isIdentifierChar(char c) {
-  return isalnum(static_cast<unsigned char>(c)) || c == '_' || c == '$' ||
-         c == '.';
-}
-
 bool AsmParser::expandMacro(raw_svector_ostream &OS, MCAsmMacro &Macro,
                             ArrayRef<MCAsmMacroParameter> Parameters,
                             ArrayRef<MCAsmMacroArgument> A,
@@ -2525,7 +2516,7 @@ bool AsmParser::expandMacro(raw_svector_ostream &OS, MCAsmMacro &Macro,
       }
 
       size_t Pos = ++I;
-      while (I != End && isIdentifierChar(Body[I]))
+      while (I != End && AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false))
         ++I;
       StringRef Argument(Body.data() + Pos, I - Pos);
       if (AltMacroMode && I != End && Body[I] == '&')
@@ -2571,13 +2562,13 @@ bool AsmParser::expandMacro(raw_svector_ostream &OS, MCAsmMacro &Macro,
       }
     }
 
-    if (!isIdentifierChar(Body[I]) || IsDarwin) {
+    if (!AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false) || IsDarwin) {
       OS << Body[I++];
       continue;
     }
 
     const size_t Start = I;
-    while (++I && isIdentifierChar(Body[I])) {
+    while (++I && AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false)) {
     }
     StringRef Token(Body.data() + Start, I - Start);
     if (AltMacroMode) {
@@ -4903,7 +4894,8 @@ void AsmParser::checkForBadMacro(SMLoc DirectiveLoc, StringRef Name,
       Pos += 2;
     } else {
       unsigned I = Pos + 1;
-      while (isIdentifierChar(Body[I]) && I + 1 != End)
+      while (AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false) &&
+             I + 1 != End)
         ++I;
 
       const char *Begin = Body.data() + Pos + 1;



More information about the llvm-commits mailing list