Rework raw identifier lexing to avoid slowing down regular identifiers. (#3855)

Instead of special-casing tokens starting with `r`, lex them as normal
identifiers, and add a special case to `#` handling to detect if the
previous token was an `r` identifier.

This roughly doubles the time to lex a raw identifier, because we do two
hash table insertions rather than one, and probably slightly slows down
lexing string literals starting with `#`, but should remove the 2%
overhead to identifier lexing from the previous approach.
This commit is contained in:
Richard Smith
2024-04-03 23:42:28 +00:00
committed by GitHub
parent 80ca6b1298
commit 45c071f2af
3 changed files with 110 additions and 45 deletions
+32 -31
View File
@@ -16,6 +16,7 @@
#include "toolchain/lex/helpers.h"
#include "toolchain/lex/numeric_literal.h"
#include "toolchain/lex/string_literal.h"
#include "toolchain/lex/token_kind.h"
#include "toolchain/lex/tokenized_buffer.h"
#if __ARM_NEON
@@ -144,8 +145,7 @@ class [[clang::internal_linkage]] Lexer {
auto LexKeywordOrIdentifier(llvm::StringRef source_text, ssize_t& position)
-> LexResult;
auto LexKeywordOrIdentifierMaybeRaw(llvm::StringRef source_text,
ssize_t& position) -> LexResult;
auto LexHash(llvm::StringRef source_text, ssize_t& position) -> LexResult;
auto LexError(llvm::StringRef source_text, ssize_t& position) -> LexResult;
@@ -472,7 +472,7 @@ static auto DispatchNext(Lexer& lexer, llvm::StringRef source_text,
CARBON_DISPATCH_LEX_TOKEN(LexError)
CARBON_DISPATCH_LEX_TOKEN(LexSymbolToken)
CARBON_DISPATCH_LEX_TOKEN(LexKeywordOrIdentifier)
CARBON_DISPATCH_LEX_TOKEN(LexKeywordOrIdentifierMaybeRaw)
CARBON_DISPATCH_LEX_TOKEN(LexHash)
CARBON_DISPATCH_LEX_TOKEN(LexNumericLiteral)
CARBON_DISPATCH_LEX_TOKEN(LexStringLiteral)
@@ -576,7 +576,6 @@ static constexpr auto MakeDispatchTable() -> DispatchTableT {
for (unsigned char c = 'a'; c <= 'z'; ++c) {
table[c] = &DispatchLexKeywordOrIdentifier;
}
table['r'] = &DispatchLexKeywordOrIdentifierMaybeRaw;
for (unsigned char c = 'A'; c <= 'Z'; ++c) {
table[c] = &DispatchLexKeywordOrIdentifier;
}
@@ -594,7 +593,7 @@ static constexpr auto MakeDispatchTable() -> DispatchTableT {
table['\''] = &DispatchLexStringLiteral;
table['"'] = &DispatchLexStringLiteral;
table['#'] = &DispatchLexStringLiteral;
table['#'] = &DispatchLexHash;
table[' '] = &DispatchLexHorizontalWhitespace;
table['\t'] = &DispatchLexHorizontalWhitespace;
@@ -1104,40 +1103,42 @@ auto Lexer::LexKeywordOrIdentifier(llvm::StringRef source_text,
.ident_id = buffer_.value_stores_->identifiers().Add(identifier_text)});
}
auto Lexer::LexKeywordOrIdentifierMaybeRaw(llvm::StringRef source_text,
ssize_t& position) -> LexResult {
CARBON_CHECK(source_text[position] == 'r');
// Raw identifiers must look like `r#<valid identifier>`, otherwise it's an
// identifier starting with the 'r'.
// TODO: Need to add support for Unicode lexing.
if (LLVM_LIKELY(position + 2 >= static_cast<ssize_t>(source_text.size()) ||
source_text[position + 1] != '#' ||
!IsIdStartByteTable[static_cast<unsigned char>(
source_text[position + 2])])) {
// TODO: Should this print a different error when there is `r#`, but it
// isn't followed by identifier text? Or is it right to put it back so
// that the `#` could be parsed as part of a raw string literal?
return LexKeywordOrIdentifier(source_text, position);
}
auto Lexer::LexHash(llvm::StringRef source_text, ssize_t& position)
-> LexResult {
// For `r#`, we already lexed an `r` identifier token. Detect that case and
// replace that token with a raw identifier. We do this to keep identifier
// lexing as fast as possible.
int column = ComputeColumn(position);
// Look for the `r` token. Note that this is always in bounds because we
// create a start of file token.
auto& prev_token_info = buffer_.token_infos_.back();
// If the previous token isn't the identifier `r`, or the character after `#`
// isn't the start of an identifier, this is not a raw identifier.
if (prev_token_info.kind != TokenKind::Identifier ||
source_text[position - 1] != 'r' ||
position + 1 == static_cast<ssize_t>(source_text.size()) ||
!IsIdStartByteTable[static_cast<unsigned char>(
source_text[position + 1])] ||
prev_token_info.token_line != current_line() ||
prev_token_info.column != ComputeColumn(position) - 1) {
[[clang::musttail]] return LexStringLiteral(source_text, position);
}
CARBON_DCHECK(buffer_.value_stores_->identifiers().Get(
prev_token_info.ident_id) == "r");
// Take the valid characters off the front of the source buffer.
llvm::StringRef identifier_text =
ScanForIdentifierPrefix(source_text.substr(position + 2));
ScanForIdentifierPrefix(source_text.substr(position + 1));
CARBON_CHECK(!identifier_text.empty()) << "Must have at least one character!";
position += identifier_text.size() + 2;
position += 1 + identifier_text.size();
// Versus LexKeywordOrIdentifier, raw identifiers do not do keyword checks.
// Otherwise we have a raw identifier.
// Replace the `r` identifier's value with the raw identifier.
// TODO: This token doesn't carry any indicator that it's raw, so
// diagnostics are unclear.
return buffer_.AddToken(
{.kind = TokenKind::Identifier,
.token_line = current_line(),
.column = column,
.ident_id = buffer_.value_stores_->identifiers().Add(identifier_text)});
prev_token_info.ident_id =
buffer_.value_stores_->identifiers().Add(identifier_text);
return LexResult(TokenIndex(buffer_.token_infos_.size() - 1));
}
auto Lexer::LexError(llvm::StringRef source_text, ssize_t& position)