[llvm] [MCParser] Deduplicate isIdentifierChar. (#218100) (PR #218851)

via llvm-commits llvm-commits at lists.llvm.org
Wed Aug 26 00:14:58 PDT 2026


llvmorg-github-actions[bot] wrote:


<!--LLVM PR SUMMARY COMMENT-->

@llvm/pr-subscribers-llvm-mc

Author: original-cooling-space (zhangweize9-cyber)

<details>
<summary>Changes</summary>

This PR supersedes #<!-- -->218100.

To address the performance regression caused by disabling inlining for a frequently called "hot" function—an issue raised regarding the previous pull request—I moved the `isIdentifierChar` function from the `AsmLexer` module into the `AsmLexer` header file and applied the `LLVM_ATTRIBUTE_ALWAYS_INLINE` attribute to force it to be inlined.

Additionally, to address the issue of relaxed string restrictions, I have set the default values ​​for `AllowAt` and `AllowHash` to `false`.

---
Full diff: https://github.com/llvm/llvm-project/pull/218851.diff


3 Files Affected:

- (modified) llvm/include/llvm/MC/MCParser/AsmLexer.h (+8) 
- (modified) llvm/lib/MC/MCParser/AsmLexer.cpp (-6) 
- (modified) llvm/lib/MC/MCParser/AsmParser.cpp (+5-13) 


``````````diff
diff --git a/llvm/include/llvm/MC/MCParser/AsmLexer.h b/llvm/include/llvm/MC/MCParser/AsmLexer.h
index 7848fc706d5eb..6b7cdb8d7822f 100644
--- a/llvm/include/llvm/MC/MCParser/AsmLexer.h
+++ b/llvm/include/llvm/MC/MCParser/AsmLexer.h
@@ -15,6 +15,7 @@
 
 #include "llvm/ADT/ArrayRef.h"
 #include "llvm/ADT/SmallVector.h"
+#include "llvm/ADT/StringExtras.h"
 #include "llvm/ADT/StringRef.h"
 #include "llvm/MC/MCAsmMacro.h"
 #include "llvm/Support/Compiler.h"
@@ -130,6 +131,13 @@ class AsmLexer {
     return Tok;
   }
 
+  /// LexIdentifier: [a-zA-Z_$.@?][a-zA-Z0-9_$.@#?]*
+  static LLVM_ATTRIBUTE_ALWAYS_INLINE
+  bool isIdentifierChar(char C, bool AllowAt = false, bool AllowHash = false) {
+    return llvm::isAlnum(C) || C == '_' || C == '$' || C == '.' || C == '?' ||
+           (AllowAt && C == '@') || (AllowHash && C == '#');
+  }
+
   /// Look ahead an arbitrary number of tokens.
   LLVM_ABI size_t peekTokens(MutableArrayRef<AsmToken> Buf,
                              bool ShouldSkipSpace = true);
diff --git a/llvm/lib/MC/MCParser/AsmLexer.cpp b/llvm/lib/MC/MCParser/AsmLexer.cpp
index 8e4b7be98bdb6..cb7bb903276d4 100644
--- a/llvm/lib/MC/MCParser/AsmLexer.cpp
+++ b/llvm/lib/MC/MCParser/AsmLexer.cpp
@@ -228,12 +228,6 @@ AsmToken AsmLexer::LexHexFloatLiteral(bool NoIntDigits) {
   return AsmToken(AsmToken::Real, StringRef(TokStart, CurPtr - TokStart));
 }
 
-/// LexIdentifier: [a-zA-Z_$.@?][a-zA-Z0-9_$.@#?]*
-static bool isIdentifierChar(char C, bool AllowAt, bool AllowHash) {
-  return isAlnum(C) || C == '_' || C == '$' || C == '.' || C == '?' ||
-         (AllowAt && C == '@') || (AllowHash && C == '#');
-}
-
 AsmToken AsmLexer::LexIdentifier() {
   // Check for floating point literals.
   if (CurPtr[-1] == '.' && isDigit(*CurPtr)) {
diff --git a/llvm/lib/MC/MCParser/AsmParser.cpp b/llvm/lib/MC/MCParser/AsmParser.cpp
index 320a3d896e9d0..610bfdbe1de89 100644
--- a/llvm/lib/MC/MCParser/AsmParser.cpp
+++ b/llvm/lib/MC/MCParser/AsmParser.cpp
@@ -2459,15 +2459,6 @@ void AsmParser::DiagHandler(const SMDiagnostic &Diag, void *Context) {
     Parser->getContext().diagnose(NewDiag);
 }
 
-// FIXME: This is mostly duplicated from the function in AsmLexer.cpp. The
-// difference being that that function accepts '@' as part of identifiers and
-// we can't do that. AsmLexer.cpp should probably be changed to handle
-// '@' as a special case when needed.
-static bool isIdentifierChar(char c) {
-  return isalnum(static_cast<unsigned char>(c)) || c == '_' || c == '$' ||
-         c == '.';
-}
-
 bool AsmParser::expandMacro(raw_svector_ostream &OS, MCAsmMacro &Macro,
                             ArrayRef<MCAsmMacroParameter> Parameters,
                             ArrayRef<MCAsmMacroArgument> A,
@@ -2525,7 +2516,7 @@ bool AsmParser::expandMacro(raw_svector_ostream &OS, MCAsmMacro &Macro,
       }
 
       size_t Pos = ++I;
-      while (I != End && isIdentifierChar(Body[I]))
+      while (I != End && AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false))
         ++I;
       StringRef Argument(Body.data() + Pos, I - Pos);
       if (AltMacroMode && I != End && Body[I] == '&')
@@ -2571,13 +2562,13 @@ bool AsmParser::expandMacro(raw_svector_ostream &OS, MCAsmMacro &Macro,
       }
     }
 
-    if (!isIdentifierChar(Body[I]) || IsDarwin) {
+    if (!AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false) || IsDarwin) {
       OS << Body[I++];
       continue;
     }
 
     const size_t Start = I;
-    while (++I && isIdentifierChar(Body[I])) {
+    while (++I && AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false)) {
     }
     StringRef Token(Body.data() + Start, I - Start);
     if (AltMacroMode) {
@@ -4903,7 +4894,8 @@ void AsmParser::checkForBadMacro(SMLoc DirectiveLoc, StringRef Name,
       Pos += 2;
     } else {
       unsigned I = Pos + 1;
-      while (isIdentifierChar(Body[I]) && I + 1 != End)
+      while (AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false) &&
+             I + 1 != End)
         ++I;
 
       const char *Begin = Body.data() + Pos + 1;

``````````

</details>


https://github.com/llvm/llvm-project/pull/218851


More information about the llvm-commits mailing list