From 5c2338dbd051b11e9e49fdcd68f60a970e874ed1 Mon Sep 17 00:00:00 2001 From: Chandler Carruth Date: Tue, 22 Aug 2023 15:55:06 -0700 Subject: [PATCH] Introduce a full lexer benchmark. (#3121) Currently this is focused on benchmarking the identifier and token lexing paths, but should expand in the future to cover other parts of the lexer. In order to effectively benchmark tokens, this adds support to the `token_kind` library to produce a list of all the tokens in Carbon that the benchmark can use to create random inputs. For identifiers, the benchmark has support for benchmarking different distributions of identifier sizes so it is easy to zoom into the performance specifically of short or long identifiers. This benchmarking is motivated by profiling overall toolchain performance and noticing that an unreasonable amount of time is spent in the lexer. In turn, the identifier lexing was surprisingly hot. I have performance improvements in the works following this, but wanted to separately introduce the benchmarking framework as the review focus will be completely different. --------- Co-authored-by: Richard Smith Co-authored-by: Jon Ross-Perkins --- toolchain/lexer/BUILD | 15 ++ toolchain/lexer/token_kind.h | 14 ++ .../lexer/tokenized_buffer_benchmark.cpp | 159 ++++++++++++++++++ 3 files changed, 188 insertions(+) create mode 100644 toolchain/lexer/tokenized_buffer_benchmark.cpp diff --git a/toolchain/lexer/BUILD b/toolchain/lexer/BUILD index ec26606d1b96..0b64e3cfb2ed 100644 --- a/toolchain/lexer/BUILD +++ b/toolchain/lexer/BUILD @@ -233,6 +233,21 @@ cc_fuzz_test( ], ) +cc_binary( + name = "tokenized_buffer_benchmark", + testonly = 1, + srcs = ["tokenized_buffer_benchmark.cpp"], + deps = [ + ":token_kind", + ":tokenized_buffer", + "//common:check", + "//toolchain/diagnostics:null_diagnostics", + "@com_github_google_benchmark//:benchmark_main", + "@com_google_absl//absl/random", + "@llvm-project//llvm:Support", + ], +) + file_test( name = "lexer_file_test", srcs = ["lexer_file_test.cpp"], diff --git a/toolchain/lexer/token_kind.h b/toolchain/lexer/token_kind.h index 1df50e188b07..4e1dfc618696 100644 --- a/toolchain/lexer/token_kind.h +++ b/toolchain/lexer/token_kind.h @@ -8,6 +8,7 @@ #include #include "common/enum_base.h" +#include "llvm/ADT/ArrayRef.h" #include "llvm/Support/FormatVariadicDetails.h" namespace Carbon { @@ -22,6 +23,9 @@ class TokenKind : public CARBON_ENUM_BASE(TokenKind) { #define CARBON_TOKEN(TokenName) CARBON_ENUM_CONSTANT_DECLARATION(TokenName) #include "toolchain/lexer/token_kind.def" + // An array of all the keyword tokens. + static const llvm::ArrayRef KeywordTokens; + // Test whether this kind of token is a simple symbol sequence (punctuation, // not letters) that appears directly in the source text and can be // unambiguously lexed with `starts_with` logic. While these may appear @@ -74,12 +78,22 @@ class TokenKind : public CARBON_ENUM_BASE(TokenKind) { } return false; } + + private: + static const TokenKind KeywordTokensStorage[]; }; #define CARBON_TOKEN(TokenName) \ CARBON_ENUM_CONSTANT_DEFINITION(TokenKind, TokenName) #include "toolchain/lexer/token_kind.def" +constexpr TokenKind TokenKind::KeywordTokensStorage[] = { +#define CARBON_KEYWORD_TOKEN(TokenName, Spelling) TokenKind::TokenName, +#include "toolchain/lexer/token_kind.def" +}; +constexpr llvm::ArrayRef TokenKind::KeywordTokens = + KeywordTokensStorage; + } // namespace Carbon namespace llvm { diff --git a/toolchain/lexer/tokenized_buffer_benchmark.cpp b/toolchain/lexer/tokenized_buffer_benchmark.cpp new file mode 100644 index 000000000000..4ec8a4f94f0f --- /dev/null +++ b/toolchain/lexer/tokenized_buffer_benchmark.cpp @@ -0,0 +1,159 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception + +#include + +#include "absl/random/random.h" +#include "common/check.h" +#include "llvm/ADT/Sequence.h" +#include "llvm/ADT/StringExtras.h" +#include "toolchain/diagnostics/null_diagnostics.h" +#include "toolchain/lexer/token_kind.h" +#include "toolchain/lexer/tokenized_buffer.h" + +namespace Carbon::Testing { +namespace { + +class LexerBenchHelper { + public: + explicit LexerBenchHelper(llvm::StringRef text) + : source_(MakeSourceBuffer(text)) {} + + auto Lex() -> TokenizedBuffer { + DiagnosticConsumer& consumer = NullDiagnosticConsumer(); + return TokenizedBuffer::Lex(source_, consumer); + } + + auto DiagnoseErrors() -> std::string { + std::string result; + llvm::raw_string_ostream out(result); + StreamDiagnosticConsumer consumer(out); + auto buffer = TokenizedBuffer::Lex(source_, consumer); + consumer.Flush(); + CARBON_CHECK(buffer.has_errors()) + << "Asked to diagnose errors but none found!"; + return result; + } + + private: + auto MakeSourceBuffer(llvm::StringRef text) -> SourceBuffer { + CARBON_CHECK(fs_.addFile(filename_, /*ModificationTime=*/0, + llvm::MemoryBuffer::getMemBuffer(text))); + return std::move(*SourceBuffer::CreateFromFile(fs_, filename_)); + } + + llvm::vfs::InMemoryFileSystem fs_; + std::string filename_ = "test.carbon"; + SourceBuffer source_; +}; + +// A large value for measurement stability without making benchmarking too slow. +constexpr int NumTokens = 100000; + +void BM_ValidKeywords(benchmark::State& state) { + absl::BitGen gen; + std::string source; + llvm::raw_string_ostream os(source); + llvm::ListSeparator sep(" "); + for (int i : llvm::seq(0, NumTokens)) { + static_cast(i); + int token = absl::Uniform(gen, 0, TokenKind::KeywordTokens.size()); + os << sep << TokenKind::KeywordTokens[token].fixed_spelling(); + } + + LexerBenchHelper helper(source); + for (auto _ : state) { + TokenizedBuffer buffer = helper.Lex(); + CARBON_CHECK(!buffer.has_errors()); + } + + state.counters["TokenRate"] = benchmark::Counter( + NumTokens, benchmark::Counter::kIsIterationInvariantRate); +} +BENCHMARK(BM_ValidKeywords); + +auto IdentifierStartChars() -> llvm::ArrayRef { + static llvm::SmallVector chars = [] { + llvm::SmallVector chars; + chars.push_back('_'); + for (char c : llvm::seq_inclusive('A', 'Z')) { + chars.push_back(c); + } + for (char c : llvm::seq_inclusive('a', 'z')) { + chars.push_back(c); + } + return chars; + }(); + return chars; +} + +auto IdentifierChars() -> llvm::ArrayRef { + static llvm::SmallVector chars = [] { + llvm::ArrayRef start_chars = IdentifierStartChars(); + llvm::SmallVector chars(start_chars.begin(), start_chars.end()); + for (char c : llvm::seq_inclusive('0', '9')) { + chars.push_back(c); + } + return chars; + }(); + return chars; +} + +void BM_ValidIdentifiers(benchmark::State& state) { + int min_length = state.range(0); + int max_length = state.range(1); + absl::BitGen gen; + std::string source; + llvm::raw_string_ostream os(source); + llvm::ListSeparator sep(" "); + for (int i : llvm::seq(0, NumTokens)) { + static_cast(i); + int length = absl::Uniform(absl::IntervalClosedClosed, gen, min_length, + max_length); + os << sep; + int id_start = source.size(); + llvm::StringRef id; + do { + // Erase any prior attempts to find an identifier. + source.resize(id_start); + llvm::ArrayRef start_chars = IdentifierStartChars(); + os << start_chars[absl::Uniform(gen, 0, start_chars.size())]; + llvm::ArrayRef chars = IdentifierChars(); + for (int j : llvm::seq(0, length)) { + static_cast(j); + os << chars[absl::Uniform(gen, 0, chars.size())]; + } + // Check if we ended up forming an integer type literal or a keyword, and + // try again. + id = llvm::StringRef(source).substr(id_start); + } while (llvm::any_of( + TokenKind::KeywordTokens, + [id](auto token) { return id == token.fixed_spelling(); }) || + ((id.consume_front("i") || id.consume_front("u") || + id.consume_front("f")) && + llvm::all_of(id, [](const char c) { return llvm::isDigit(c); }))); + } + + LexerBenchHelper helper(source); + for (auto _ : state) { + TokenizedBuffer buffer = helper.Lex(); + + // Ensure that lexing actually occurs for benchmarking and that it doesn't + // hit errors that would skew the benchmark results. + CARBON_CHECK(!buffer.has_errors()) << helper.DiagnoseErrors(); + } + + state.counters["TokenRate"] = benchmark::Counter( + NumTokens, benchmark::Counter::kIsIterationInvariantRate); +} +BENCHMARK(BM_ValidIdentifiers) + // Benchmark a few ranges of identifier widths to cover different patterns + // that emerge with small, medium, and longer identifiers. + ->Args({1, 3}) + ->Args({3, 5}) + ->Args({3, 16}) + ->Args({12, 64}); + +} // namespace +} // namespace Carbon::Testing