mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-10-05 19:01:07 +01:00
Add partial raw identifier support. (#3344)
I'm looking at this due to the conversation on #3341. Although diagnostics aren't where they should be, I thought it may help to start adding raw identifier support (which may also help show how I was thinking about this). Note regarding the TODO on how to form the token, `GetTokenText` returns the `string_id`'s reference value for an `Identifier`. So to make `GetTokenText` work in a way that returns `r#foo` for a raw identifier, I think there are a few options: 1. Add additional data indicating the end of the identifier. 2. Add `RawIdentifier` as a token kind to indicate that it's raw and should be prefixed with `r#` (but also giving later stages one more token kind to handle) 3. Make the `string_id` correspond to `r#foo`, and have later stages add `foo` to the strings table whenever `r#foo` is encountered (with map lookups leading to deduplication). 4. Add `StringId::RawKeyword` special values for each keyword. - This would mean `self` prints as `self`, `r#self` prints as `r#self`, but `r#foo` is not a keyword so prints as `foo`. - This means keywords would need to be listed in a place `StringId` can depend on them, one way or the other (e.g., a `keywords.def` file in `base/` should work). 5. Say that it _is_ an `Identifier`, and if it's a keyword spelling, it must have been a raw identifier. - Same limitation as above: This would mean `self` prints as `self`, `r#self` prints as `r#self`, but `r#foo` is not a keyword so prints as `foo`. I'm hoping to resolve this issue separately though. :)
This commit is contained in:
@@ -427,6 +427,66 @@ void BM_ValidKeywords(benchmark::State& state) {
|
||||
}
|
||||
BENCHMARK(BM_ValidKeywords);
|
||||
|
||||
void BM_ValidKeywordsAsRawIdentifiers(benchmark::State& state) {
|
||||
absl::BitGen gen;
|
||||
std::array<llvm::StringRef, NumTokens> tokens;
|
||||
for (int i : llvm::seq(NumTokens)) {
|
||||
tokens[i] = TokenKind::KeywordTokens[i % TokenKind::KeywordTokens.size()]
|
||||
.fixed_spelling();
|
||||
}
|
||||
std::shuffle(tokens.begin(), tokens.end(), gen);
|
||||
std::string source("r#");
|
||||
source.append(llvm::join(tokens, " r#"));
|
||||
|
||||
LexerBenchHelper helper(source);
|
||||
for (auto _ : state) {
|
||||
TokenizedBuffer buffer = helper.Lex();
|
||||
CARBON_CHECK(!buffer.has_errors());
|
||||
}
|
||||
|
||||
state.SetBytesProcessed(state.iterations() * source.size());
|
||||
state.counters["tokens_per_second"] = benchmark::Counter(
|
||||
NumTokens, benchmark::Counter::kIsIterationInvariantRate);
|
||||
}
|
||||
BENCHMARK(BM_ValidKeywordsAsRawIdentifiers);
|
||||
|
||||
// This benchmark does a 50-50 split of r-prefixed and r#-prefixed identifiers
|
||||
// to directly compare raw and non-raw performance.
|
||||
void BM_RawIdentifierFocus(benchmark::State& state) {
|
||||
const std::array<std::string, NumTokens>& ids = GetRandomIdentifiers();
|
||||
|
||||
llvm::SmallVector<std::string> modified_ids;
|
||||
// As we resize, start with the in-use prefix. Note that `r#` uses the first
|
||||
// character of the original identifier.
|
||||
modified_ids.resize(NumTokens / 2, "r#");
|
||||
modified_ids.resize(NumTokens, "r");
|
||||
for (int i : llvm::seq(NumTokens / 2)) {
|
||||
// Use the same identifier both ways.
|
||||
modified_ids[i].append(ids[i]);
|
||||
modified_ids[i + NumTokens / 2].append(
|
||||
llvm::StringRef(ids[i]).drop_front());
|
||||
}
|
||||
|
||||
absl::BitGen gen;
|
||||
std::array<llvm::StringRef, NumTokens> tokens;
|
||||
for (int i : llvm::seq(NumTokens)) {
|
||||
tokens[i] = modified_ids[i];
|
||||
}
|
||||
std::shuffle(tokens.begin(), tokens.end(), gen);
|
||||
std::string source = llvm::join(tokens, " ");
|
||||
|
||||
LexerBenchHelper helper(source);
|
||||
for (auto _ : state) {
|
||||
TokenizedBuffer buffer = helper.Lex();
|
||||
CARBON_CHECK(!buffer.has_errors());
|
||||
}
|
||||
|
||||
state.SetBytesProcessed(state.iterations() * source.size());
|
||||
state.counters["tokens_per_second"] = benchmark::Counter(
|
||||
NumTokens, benchmark::Counter::kIsIterationInvariantRate);
|
||||
}
|
||||
BENCHMARK(BM_RawIdentifierFocus);
|
||||
|
||||
template <int MinLength, int MaxLength, bool Uniform>
|
||||
void BM_ValidIdentifiers(benchmark::State& state) {
|
||||
std::string source = RandomIdentifierSeq<MinLength, MaxLength, Uniform>();
|
||||
|
||||
Reference in New Issue
Block a user