Add partial raw identifier support. (#3344)

I'm looking at this due to the conversation on #3341. Although
diagnostics aren't where they should be, I thought it may help to start
adding raw identifier support (which may also help show how I was
thinking about this).

Note regarding the TODO on how to form the token, `GetTokenText` returns
the `string_id`'s reference value for an `Identifier`. So to make
`GetTokenText` work in a way that returns `r#foo` for a raw identifier,
I think there are a few options:

1. Add additional data indicating the end of the identifier.
2. Add `RawIdentifier` as a token kind to indicate that it's raw and
should be prefixed with `r#` (but also giving later stages one more
token kind to handle)
3. Make the `string_id` correspond to `r#foo`, and have later stages add
`foo` to the strings table whenever `r#foo` is encountered (with map
lookups leading to deduplication).
4. Add `StringId::RawKeyword` special values for each keyword.
- This would mean `self` prints as `self`, `r#self` prints as `r#self`,
but `r#foo` is not a keyword so prints as `foo`.
- This means keywords would need to be listed in a place `StringId` can
depend on them, one way or the other (e.g., a `keywords.def` file in
`base/` should work).
5. Say that it _is_ an `Identifier`, and if it's a keyword spelling, it
must have been a raw identifier.
- Same limitation as above: This would mean `self` prints as `self`,
`r#self` prints as `r#self`, but `r#foo` is not a keyword so prints as
`foo`.

I'm hoping to resolve this issue separately though. :)
This commit is contained in:
Jon Ross-Perkins
2023-10-30 18:11:55 +00:00
committed by GitHub
parent d42c1a3a7c
commit 6742d0d048
5 changed files with 220 additions and 8 deletions
+52 -8
View File
@@ -81,13 +81,9 @@ static constexpr SIMDMaskArrayT PrefixMasks = []() constexpr {
#endif // CARBON_USE_SIMD
// A table of booleans that we can use to classify bytes as being valid
// identifier (or keyword) characters. This is used in the generic,
// non-vectorized fallback code to scan for length of an identifier.
constexpr std::array<bool, 256> IsIdByteTable = [] {
// identifier start. This is used by raw identifier detection.
constexpr std::array<bool, 256> IsIdStartByteTable = [] {
std::array<bool, 256> table = {};
for (char c = '0'; c <= '9'; ++c) {
table[c] = true;
}
for (char c = 'A'; c <= 'Z'; ++c) {
table[c] = true;
}
@@ -98,6 +94,17 @@ constexpr std::array<bool, 256> IsIdByteTable = [] {
return table;
}();
// A table of booleans that we can use to classify bytes as being valid
// identifier (or keyword) characters. This is used in the generic,
// non-vectorized fallback code to scan for length of an identifier.
constexpr std::array<bool, 256> IsIdByteTable = [] {
std::array<bool, 256> table = IsIdStartByteTable;
for (char c = '0'; c <= '9'; ++c) {
table[c] = true;
}
return table;
}();
// Baseline scalar version, also available for scalar-fallback in SIMD code.
// Uses `ssize_t` for performance when indexing in the loop.
//
@@ -859,8 +866,7 @@ class [[clang::internal_linkage]] TokenizedBuffer::Lexer {
// TODO: Need to add support for Unicode lexing.
return LexError(source_text, position);
}
CARBON_CHECK(IsAlpha(source_text[position]) ||
source_text[position] == '_');
CARBON_CHECK(IsIdStartByteTable[source_text[position]]);
int column = ComputeColumn(position);
@@ -894,6 +900,42 @@ class [[clang::internal_linkage]] TokenizedBuffer::Lexer {
.string_id = buffer_.value_stores_->strings().Add(identifier_text)});
}
auto LexKeywordOrIdentifierMaybeRaw(llvm::StringRef source_text,
ssize_t& position) -> LexResult {
CARBON_CHECK(source_text[position] == 'r');
// Raw identifiers must look like `r#<valid identifier>`, otherwise it's an
// identifier starting with the 'r'.
// TODO: Need to add support for Unicode lexing.
if (LLVM_LIKELY(position + 2 >= static_cast<ssize_t>(source_text.size()) ||
source_text[position + 1] != '#' ||
!IsIdStartByteTable[source_text[position + 2]])) {
// TODO: Should this print a different error when there is `r#`, but it
// isn't followed by identifier text? Or is it right to put it back so
// that the `#` could be parsed as part of a raw string literal?
return LexKeywordOrIdentifier(source_text, position);
}
int column = ComputeColumn(position);
// Take the valid characters off the front of the source buffer.
llvm::StringRef identifier_text =
ScanForIdentifierPrefix(source_text.substr(position + 2));
CARBON_CHECK(!identifier_text.empty())
<< "Must have at least one character!";
position += identifier_text.size() + 2;
// Versus LexKeywordOrIdentifier, raw identifiers do not do keyword checks.
// Otherwise we have a raw identifier.
// TODO: This token doesn't carry any indicator that it's raw, so
// diagnostics are unclear.
return buffer_.AddToken(
{.kind = TokenKind::Identifier,
.token_line = current_line(),
.column = column,
.string_id = buffer_.value_stores_->strings().Add(identifier_text)});
}
auto LexError(llvm::StringRef source_text, ssize_t& position) -> LexResult {
llvm::StringRef error_text =
source_text.substr(position).take_while([](char c) {
@@ -1018,6 +1060,7 @@ class [[clang::internal_linkage]] TokenizedBuffer::Lexer {
CARBON_DISPATCH_LEX_TOKEN(LexError)
CARBON_DISPATCH_LEX_TOKEN(LexSymbolToken)
CARBON_DISPATCH_LEX_TOKEN(LexKeywordOrIdentifier)
CARBON_DISPATCH_LEX_TOKEN(LexKeywordOrIdentifierMaybeRaw)
CARBON_DISPATCH_LEX_TOKEN(LexNumericLiteral)
CARBON_DISPATCH_LEX_TOKEN(LexStringLiteral)
@@ -1147,6 +1190,7 @@ class [[clang::internal_linkage]] TokenizedBuffer::Lexer {
for (unsigned char c = 'a'; c <= 'z'; ++c) {
table[c] = &DispatchLexKeywordOrIdentifier;
}
table['r'] = &DispatchLexKeywordOrIdentifierMaybeRaw;
for (unsigned char c = 'A'; c <= 'Z'; ++c) {
table[c] = &DispatchLexKeywordOrIdentifier;
}