[llvm] [MCParser] Deduplicate isIdentifierChar. (#218100) (PR #218851)
via llvm-commits
llvm-commits at lists.llvm.org
Wed Aug 26 00:14:58 PDT 2026
llvmorg-github-actions[bot] wrote:
<!--LLVM PR SUMMARY COMMENT-->
@llvm/pr-subscribers-llvm-mc
Author: original-cooling-space (zhangweize9-cyber)
<details>
<summary>Changes</summary>
This PR supersedes #<!-- -->218100.
To address the performance regression caused by disabling inlining for a frequently called "hot" function—an issue raised regarding the previous pull request—I moved the `isIdentifierChar` function from the `AsmLexer` module into the `AsmLexer` header file and applied the `LLVM_ATTRIBUTE_ALWAYS_INLINE` attribute to force it to be inlined.
Additionally, to address the issue of relaxed string restrictions, I have set the default values for `AllowAt` and `AllowHash` to `false`.
---
Full diff: https://github.com/llvm/llvm-project/pull/218851.diff
3 Files Affected:
- (modified) llvm/include/llvm/MC/MCParser/AsmLexer.h (+8)
- (modified) llvm/lib/MC/MCParser/AsmLexer.cpp (-6)
- (modified) llvm/lib/MC/MCParser/AsmParser.cpp (+5-13)
``````````diff
diff --git a/llvm/include/llvm/MC/MCParser/AsmLexer.h b/llvm/include/llvm/MC/MCParser/AsmLexer.h
index 7848fc706d5eb..6b7cdb8d7822f 100644
--- a/llvm/include/llvm/MC/MCParser/AsmLexer.h
+++ b/llvm/include/llvm/MC/MCParser/AsmLexer.h
@@ -15,6 +15,7 @@
#include "llvm/ADT/ArrayRef.h"
#include "llvm/ADT/SmallVector.h"
+#include "llvm/ADT/StringExtras.h"
#include "llvm/ADT/StringRef.h"
#include "llvm/MC/MCAsmMacro.h"
#include "llvm/Support/Compiler.h"
@@ -130,6 +131,13 @@ class AsmLexer {
return Tok;
}
+ /// LexIdentifier: [a-zA-Z_$.@?][a-zA-Z0-9_$.@#?]*
+ static LLVM_ATTRIBUTE_ALWAYS_INLINE
+ bool isIdentifierChar(char C, bool AllowAt = false, bool AllowHash = false) {
+ return llvm::isAlnum(C) || C == '_' || C == '$' || C == '.' || C == '?' ||
+ (AllowAt && C == '@') || (AllowHash && C == '#');
+ }
+
/// Look ahead an arbitrary number of tokens.
LLVM_ABI size_t peekTokens(MutableArrayRef<AsmToken> Buf,
bool ShouldSkipSpace = true);
diff --git a/llvm/lib/MC/MCParser/AsmLexer.cpp b/llvm/lib/MC/MCParser/AsmLexer.cpp
index 8e4b7be98bdb6..cb7bb903276d4 100644
--- a/llvm/lib/MC/MCParser/AsmLexer.cpp
+++ b/llvm/lib/MC/MCParser/AsmLexer.cpp
@@ -228,12 +228,6 @@ AsmToken AsmLexer::LexHexFloatLiteral(bool NoIntDigits) {
return AsmToken(AsmToken::Real, StringRef(TokStart, CurPtr - TokStart));
}
-/// LexIdentifier: [a-zA-Z_$.@?][a-zA-Z0-9_$.@#?]*
-static bool isIdentifierChar(char C, bool AllowAt, bool AllowHash) {
- return isAlnum(C) || C == '_' || C == '$' || C == '.' || C == '?' ||
- (AllowAt && C == '@') || (AllowHash && C == '#');
-}
-
AsmToken AsmLexer::LexIdentifier() {
// Check for floating point literals.
if (CurPtr[-1] == '.' && isDigit(*CurPtr)) {
diff --git a/llvm/lib/MC/MCParser/AsmParser.cpp b/llvm/lib/MC/MCParser/AsmParser.cpp
index 320a3d896e9d0..610bfdbe1de89 100644
--- a/llvm/lib/MC/MCParser/AsmParser.cpp
+++ b/llvm/lib/MC/MCParser/AsmParser.cpp
@@ -2459,15 +2459,6 @@ void AsmParser::DiagHandler(const SMDiagnostic &Diag, void *Context) {
Parser->getContext().diagnose(NewDiag);
}
-// FIXME: This is mostly duplicated from the function in AsmLexer.cpp. The
-// difference being that that function accepts '@' as part of identifiers and
-// we can't do that. AsmLexer.cpp should probably be changed to handle
-// '@' as a special case when needed.
-static bool isIdentifierChar(char c) {
- return isalnum(static_cast<unsigned char>(c)) || c == '_' || c == '$' ||
- c == '.';
-}
-
bool AsmParser::expandMacro(raw_svector_ostream &OS, MCAsmMacro &Macro,
ArrayRef<MCAsmMacroParameter> Parameters,
ArrayRef<MCAsmMacroArgument> A,
@@ -2525,7 +2516,7 @@ bool AsmParser::expandMacro(raw_svector_ostream &OS, MCAsmMacro &Macro,
}
size_t Pos = ++I;
- while (I != End && isIdentifierChar(Body[I]))
+ while (I != End && AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false))
++I;
StringRef Argument(Body.data() + Pos, I - Pos);
if (AltMacroMode && I != End && Body[I] == '&')
@@ -2571,13 +2562,13 @@ bool AsmParser::expandMacro(raw_svector_ostream &OS, MCAsmMacro &Macro,
}
}
- if (!isIdentifierChar(Body[I]) || IsDarwin) {
+ if (!AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false) || IsDarwin) {
OS << Body[I++];
continue;
}
const size_t Start = I;
- while (++I && isIdentifierChar(Body[I])) {
+ while (++I && AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false)) {
}
StringRef Token(Body.data() + Start, I - Start);
if (AltMacroMode) {
@@ -4903,7 +4894,8 @@ void AsmParser::checkForBadMacro(SMLoc DirectiveLoc, StringRef Name,
Pos += 2;
} else {
unsigned I = Pos + 1;
- while (isIdentifierChar(Body[I]) && I + 1 != End)
+ while (AsmLexer::isIdentifierChar(Body[I], /*AllowAt=*/false) &&
+ I + 1 != End)
++I;
const char *Begin = Body.data() + Pos + 1;
``````````
</details>
https://github.com/llvm/llvm-project/pull/218851
More information about the llvm-commits
mailing list