diff --git a/toolchain/lexer/BUILD b/toolchain/lexer/BUILD index ec26606d1b96..0b64e3cfb2ed 100644 --- a/toolchain/lexer/BUILD +++ b/toolchain/lexer/BUILD @@ -233,6 +233,21 @@ cc_fuzz_test( ], ) +cc_binary( + name = "tokenized_buffer_benchmark", + testonly = 1, + srcs = ["tokenized_buffer_benchmark.cpp"], + deps = [ + ":token_kind", + ":tokenized_buffer", + "//common:check", + "//toolchain/diagnostics:null_diagnostics", + "@com_github_google_benchmark//:benchmark_main", + "@com_google_absl//absl/random", + "@llvm-project//llvm:Support", + ], +) + file_test( name = "lexer_file_test", srcs = ["lexer_file_test.cpp"], diff --git a/toolchain/lexer/token_kind.h b/toolchain/lexer/token_kind.h index 1df50e188b07..4e1dfc618696 100644 --- a/toolchain/lexer/token_kind.h +++ b/toolchain/lexer/token_kind.h @@ -8,6 +8,7 @@ #include #include "common/enum_base.h" +#include "llvm/ADT/ArrayRef.h" #include "llvm/Support/FormatVariadicDetails.h" namespace Carbon { @@ -22,6 +23,9 @@ class TokenKind : public CARBON_ENUM_BASE(TokenKind) { #define CARBON_TOKEN(TokenName) CARBON_ENUM_CONSTANT_DECLARATION(TokenName) #include "toolchain/lexer/token_kind.def" + // An array of all the keyword tokens. + static const llvm::ArrayRef KeywordTokens; + // Test whether this kind of token is a simple symbol sequence (punctuation, // not letters) that appears directly in the source text and can be // unambiguously lexed with `starts_with` logic. While these may appear @@ -74,12 +78,22 @@ class TokenKind : public CARBON_ENUM_BASE(TokenKind) { } return false; } + + private: + static const TokenKind KeywordTokensStorage[]; }; #define CARBON_TOKEN(TokenName) \ CARBON_ENUM_CONSTANT_DEFINITION(TokenKind, TokenName) #include "toolchain/lexer/token_kind.def" +constexpr TokenKind TokenKind::KeywordTokensStorage[] = { +#define CARBON_KEYWORD_TOKEN(TokenName, Spelling) TokenKind::TokenName, +#include "toolchain/lexer/token_kind.def" +}; +constexpr llvm::ArrayRef TokenKind::KeywordTokens = + KeywordTokensStorage; + } // namespace Carbon namespace llvm { diff --git a/toolchain/lexer/tokenized_buffer_benchmark.cpp b/toolchain/lexer/tokenized_buffer_benchmark.cpp new file mode 100644 index 000000000000..4ec8a4f94f0f --- /dev/null +++ b/toolchain/lexer/tokenized_buffer_benchmark.cpp @@ -0,0 +1,159 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include + +#include "absl/random/random.h" +#include "common/check.h" +#include "llvm/ADT/Sequence.h" +#include "llvm/ADT/StringExtras.h" +#include "toolchain/diagnostics/null_diagnostics.h" +#include "toolchain/lexer/token_kind.h" +#include "toolchain/lexer/tokenized_buffer.h" + +namespace Carbon::Testing { +namespace { + +class LexerBenchHelper { + public: + explicit LexerBenchHelper(llvm::StringRef text) + : source_(MakeSourceBuffer(text)) {} + + auto Lex() -> TokenizedBuffer { + DiagnosticConsumer& consumer = NullDiagnosticConsumer(); + return TokenizedBuffer::Lex(source_, consumer); + } + + auto DiagnoseErrors() -> std::string { + std::string result; + llvm::raw_string_ostream out(result); + StreamDiagnosticConsumer consumer(out); + auto buffer = TokenizedBuffer::Lex(source_, consumer); + consumer.Flush(); + CARBON_CHECK(buffer.has_errors()) + << "Asked to diagnose errors but none found!"; + return result; + } + + private: + auto MakeSourceBuffer(llvm::StringRef text) -> SourceBuffer { + CARBON_CHECK(fs_.addFile(filename_, /*ModificationTime=*/0, + llvm::MemoryBuffer::getMemBuffer(text))); + return std::move(*SourceBuffer::CreateFromFile(fs_, filename_)); + } + + llvm::vfs::InMemoryFileSystem fs_; + std::string filename_ = "test.carbon"; + SourceBuffer source_; +}; + +// A large value for measurement stability without making benchmarking too slow. +constexpr int NumTokens = 100000; + +void BM_ValidKeywords(benchmark::State& state) { + absl::BitGen gen; + std::string source; + llvm::raw_string_ostream os(source); + llvm::ListSeparator sep(" "); + for (int i : llvm::seq(0, NumTokens)) { + static_cast(i); + int token = absl::Uniform(gen, 0, TokenKind::KeywordTokens.size()); + os << sep << TokenKind::KeywordTokens[token].fixed_spelling(); + } + + LexerBenchHelper helper(source); + for (auto _ : state) { + TokenizedBuffer buffer = helper.Lex(); + CARBON_CHECK(!buffer.has_errors()); + } + + state.counters["TokenRate"] = benchmark::Counter( + NumTokens, benchmark::Counter::kIsIterationInvariantRate); +} +BENCHMARK(BM_ValidKeywords); + +auto IdentifierStartChars() -> llvm::ArrayRef { + static llvm::SmallVector chars = [] { + llvm::SmallVector chars; + chars.push_back('_'); + for (char c : llvm::seq_inclusive('A', 'Z')) { + chars.push_back(c); + } + for (char c : llvm::seq_inclusive('a', 'z')) { + chars.push_back(c); + } + return chars; + }(); + return chars; +} + +auto IdentifierChars() -> llvm::ArrayRef { + static llvm::SmallVector chars = [] { + llvm::ArrayRef start_chars = IdentifierStartChars(); + llvm::SmallVector chars(start_chars.begin(), start_chars.end()); + for (char c : llvm::seq_inclusive('0', '9')) { + chars.push_back(c); + } + return chars; + }(); + return chars; +} + +void BM_ValidIdentifiers(benchmark::State& state) { + int min_length = state.range(0); + int max_length = state.range(1); + absl::BitGen gen; + std::string source; + llvm::raw_string_ostream os(source); + llvm::ListSeparator sep(" "); + for (int i : llvm::seq(0, NumTokens)) { + static_cast(i); + int length = absl::Uniform(absl::IntervalClosedClosed, gen, min_length, + max_length); + os << sep; + int id_start = source.size(); + llvm::StringRef id; + do { + // Erase any prior attempts to find an identifier. + source.resize(id_start); + llvm::ArrayRef start_chars = IdentifierStartChars(); + os << start_chars[absl::Uniform(gen, 0, start_chars.size())]; + llvm::ArrayRef chars = IdentifierChars(); + for (int j : llvm::seq(0, length)) { + static_cast(j); + os << chars[absl::Uniform(gen, 0, chars.size())]; + } + // Check if we ended up forming an integer type literal or a keyword, and + // try again. + id = llvm::StringRef(source).substr(id_start); + } while (llvm::any_of( + TokenKind::KeywordTokens, + [id](auto token) { return id == token.fixed_spelling(); }) || + ((id.consume_front("i") || id.consume_front("u") || + id.consume_front("f")) && + llvm::all_of(id, [](const char c) { return llvm::isDigit(c); }))); + } + + LexerBenchHelper helper(source); + for (auto _ : state) { + TokenizedBuffer buffer = helper.Lex(); + + // Ensure that lexing actually occurs for benchmarking and that it doesn't + // hit errors that would skew the benchmark results. + CARBON_CHECK(!buffer.has_errors()) << helper.DiagnoseErrors(); + } + + state.counters["TokenRate"] = benchmark::Counter( + NumTokens, benchmark::Counter::kIsIterationInvariantRate); +} +BENCHMARK(BM_ValidIdentifiers) + // Benchmark a few ranges of identifier widths to cover different patterns + // that emerge with small, medium, and longer identifiers. + ->Args({1, 3}) + ->Args({3, 5}) + ->Args({3, 16}) + ->Args({12, 64}); + +} // namespace +} // namespace Carbon::Testing