mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-10-05 22:02:55 +01:00
Introduce a full lexer benchmark. (#3121)
Currently this is focused on benchmarking the identifier and token lexing paths, but should expand in the future to cover other parts of the lexer. In order to effectively benchmark tokens, this adds support to the `token_kind` library to produce a list of all the tokens in Carbon that the benchmark can use to create random inputs. For identifiers, the benchmark has support for benchmarking different distributions of identifier sizes so it is easy to zoom into the performance specifically of short or long identifiers. This benchmarking is motivated by profiling overall toolchain performance and noticing that an unreasonable amount of time is spent in the lexer. In turn, the identifier lexing was surprisingly hot. I have performance improvements in the works following this, but wanted to separately introduce the benchmarking framework as the review focus will be completely different. --------- Co-authored-by: Richard Smith <richard@metafoo.co.uk> Co-authored-by: Jon Ross-Perkins <jperkins@google.com>
This commit is contained in:
co-authored by
Richard Smith
Jon Ross-Perkins
parent
7f4cf794cf
commit
5c2338dbd0
@@ -233,6 +233,21 @@ cc_fuzz_test(
|
||||
],
|
||||
)
|
||||
|
||||
cc_binary(
|
||||
name = "tokenized_buffer_benchmark",
|
||||
testonly = 1,
|
||||
srcs = ["tokenized_buffer_benchmark.cpp"],
|
||||
deps = [
|
||||
":token_kind",
|
||||
":tokenized_buffer",
|
||||
"//common:check",
|
||||
"//toolchain/diagnostics:null_diagnostics",
|
||||
"@com_github_google_benchmark//:benchmark_main",
|
||||
"@com_google_absl//absl/random",
|
||||
"@llvm-project//llvm:Support",
|
||||
],
|
||||
)
|
||||
|
||||
file_test(
|
||||
name = "lexer_file_test",
|
||||
srcs = ["lexer_file_test.cpp"],
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
#include <cstdint>
|
||||
|
||||
#include "common/enum_base.h"
|
||||
#include "llvm/ADT/ArrayRef.h"
|
||||
#include "llvm/Support/FormatVariadicDetails.h"
|
||||
|
||||
namespace Carbon {
|
||||
@@ -22,6 +23,9 @@ class TokenKind : public CARBON_ENUM_BASE(TokenKind) {
|
||||
#define CARBON_TOKEN(TokenName) CARBON_ENUM_CONSTANT_DECLARATION(TokenName)
|
||||
#include "toolchain/lexer/token_kind.def"
|
||||
|
||||
// An array of all the keyword tokens.
|
||||
static const llvm::ArrayRef<TokenKind> KeywordTokens;
|
||||
|
||||
// Test whether this kind of token is a simple symbol sequence (punctuation,
|
||||
// not letters) that appears directly in the source text and can be
|
||||
// unambiguously lexed with `starts_with` logic. While these may appear
|
||||
@@ -74,12 +78,22 @@ class TokenKind : public CARBON_ENUM_BASE(TokenKind) {
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
private:
|
||||
static const TokenKind KeywordTokensStorage[];
|
||||
};
|
||||
|
||||
#define CARBON_TOKEN(TokenName) \
|
||||
CARBON_ENUM_CONSTANT_DEFINITION(TokenKind, TokenName)
|
||||
#include "toolchain/lexer/token_kind.def"
|
||||
|
||||
constexpr TokenKind TokenKind::KeywordTokensStorage[] = {
|
||||
#define CARBON_KEYWORD_TOKEN(TokenName, Spelling) TokenKind::TokenName,
|
||||
#include "toolchain/lexer/token_kind.def"
|
||||
};
|
||||
constexpr llvm::ArrayRef<TokenKind> TokenKind::KeywordTokens =
|
||||
KeywordTokensStorage;
|
||||
|
||||
} // namespace Carbon
|
||||
|
||||
namespace llvm {
|
||||
|
||||
@@ -0,0 +1,159 @@
|
||||
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
|
||||
// Exceptions. See /LICENSE for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
#include <benchmark/benchmark.h>
|
||||
|
||||
#include "absl/random/random.h"
|
||||
#include "common/check.h"
|
||||
#include "llvm/ADT/Sequence.h"
|
||||
#include "llvm/ADT/StringExtras.h"
|
||||
#include "toolchain/diagnostics/null_diagnostics.h"
|
||||
#include "toolchain/lexer/token_kind.h"
|
||||
#include "toolchain/lexer/tokenized_buffer.h"
|
||||
|
||||
namespace Carbon::Testing {
|
||||
namespace {
|
||||
|
||||
class LexerBenchHelper {
|
||||
public:
|
||||
explicit LexerBenchHelper(llvm::StringRef text)
|
||||
: source_(MakeSourceBuffer(text)) {}
|
||||
|
||||
auto Lex() -> TokenizedBuffer {
|
||||
DiagnosticConsumer& consumer = NullDiagnosticConsumer();
|
||||
return TokenizedBuffer::Lex(source_, consumer);
|
||||
}
|
||||
|
||||
auto DiagnoseErrors() -> std::string {
|
||||
std::string result;
|
||||
llvm::raw_string_ostream out(result);
|
||||
StreamDiagnosticConsumer consumer(out);
|
||||
auto buffer = TokenizedBuffer::Lex(source_, consumer);
|
||||
consumer.Flush();
|
||||
CARBON_CHECK(buffer.has_errors())
|
||||
<< "Asked to diagnose errors but none found!";
|
||||
return result;
|
||||
}
|
||||
|
||||
private:
|
||||
auto MakeSourceBuffer(llvm::StringRef text) -> SourceBuffer {
|
||||
CARBON_CHECK(fs_.addFile(filename_, /*ModificationTime=*/0,
|
||||
llvm::MemoryBuffer::getMemBuffer(text)));
|
||||
return std::move(*SourceBuffer::CreateFromFile(fs_, filename_));
|
||||
}
|
||||
|
||||
llvm::vfs::InMemoryFileSystem fs_;
|
||||
std::string filename_ = "test.carbon";
|
||||
SourceBuffer source_;
|
||||
};
|
||||
|
||||
// A large value for measurement stability without making benchmarking too slow.
|
||||
constexpr int NumTokens = 100000;
|
||||
|
||||
void BM_ValidKeywords(benchmark::State& state) {
|
||||
absl::BitGen gen;
|
||||
std::string source;
|
||||
llvm::raw_string_ostream os(source);
|
||||
llvm::ListSeparator sep(" ");
|
||||
for (int i : llvm::seq(0, NumTokens)) {
|
||||
static_cast<void>(i);
|
||||
int token = absl::Uniform<int>(gen, 0, TokenKind::KeywordTokens.size());
|
||||
os << sep << TokenKind::KeywordTokens[token].fixed_spelling();
|
||||
}
|
||||
|
||||
LexerBenchHelper helper(source);
|
||||
for (auto _ : state) {
|
||||
TokenizedBuffer buffer = helper.Lex();
|
||||
CARBON_CHECK(!buffer.has_errors());
|
||||
}
|
||||
|
||||
state.counters["TokenRate"] = benchmark::Counter(
|
||||
NumTokens, benchmark::Counter::kIsIterationInvariantRate);
|
||||
}
|
||||
BENCHMARK(BM_ValidKeywords);
|
||||
|
||||
auto IdentifierStartChars() -> llvm::ArrayRef<char> {
|
||||
static llvm::SmallVector<char> chars = [] {
|
||||
llvm::SmallVector<char> chars;
|
||||
chars.push_back('_');
|
||||
for (char c : llvm::seq_inclusive('A', 'Z')) {
|
||||
chars.push_back(c);
|
||||
}
|
||||
for (char c : llvm::seq_inclusive('a', 'z')) {
|
||||
chars.push_back(c);
|
||||
}
|
||||
return chars;
|
||||
}();
|
||||
return chars;
|
||||
}
|
||||
|
||||
auto IdentifierChars() -> llvm::ArrayRef<char> {
|
||||
static llvm::SmallVector<char> chars = [] {
|
||||
llvm::ArrayRef<char> start_chars = IdentifierStartChars();
|
||||
llvm::SmallVector<char> chars(start_chars.begin(), start_chars.end());
|
||||
for (char c : llvm::seq_inclusive('0', '9')) {
|
||||
chars.push_back(c);
|
||||
}
|
||||
return chars;
|
||||
}();
|
||||
return chars;
|
||||
}
|
||||
|
||||
void BM_ValidIdentifiers(benchmark::State& state) {
|
||||
int min_length = state.range(0);
|
||||
int max_length = state.range(1);
|
||||
absl::BitGen gen;
|
||||
std::string source;
|
||||
llvm::raw_string_ostream os(source);
|
||||
llvm::ListSeparator sep(" ");
|
||||
for (int i : llvm::seq(0, NumTokens)) {
|
||||
static_cast<void>(i);
|
||||
int length = absl::Uniform<int>(absl::IntervalClosedClosed, gen, min_length,
|
||||
max_length);
|
||||
os << sep;
|
||||
int id_start = source.size();
|
||||
llvm::StringRef id;
|
||||
do {
|
||||
// Erase any prior attempts to find an identifier.
|
||||
source.resize(id_start);
|
||||
llvm::ArrayRef<char> start_chars = IdentifierStartChars();
|
||||
os << start_chars[absl::Uniform<int>(gen, 0, start_chars.size())];
|
||||
llvm::ArrayRef<char> chars = IdentifierChars();
|
||||
for (int j : llvm::seq(0, length)) {
|
||||
static_cast<void>(j);
|
||||
os << chars[absl::Uniform<int>(gen, 0, chars.size())];
|
||||
}
|
||||
// Check if we ended up forming an integer type literal or a keyword, and
|
||||
// try again.
|
||||
id = llvm::StringRef(source).substr(id_start);
|
||||
} while (llvm::any_of(
|
||||
TokenKind::KeywordTokens,
|
||||
[id](auto token) { return id == token.fixed_spelling(); }) ||
|
||||
((id.consume_front("i") || id.consume_front("u") ||
|
||||
id.consume_front("f")) &&
|
||||
llvm::all_of(id, [](const char c) { return llvm::isDigit(c); })));
|
||||
}
|
||||
|
||||
LexerBenchHelper helper(source);
|
||||
for (auto _ : state) {
|
||||
TokenizedBuffer buffer = helper.Lex();
|
||||
|
||||
// Ensure that lexing actually occurs for benchmarking and that it doesn't
|
||||
// hit errors that would skew the benchmark results.
|
||||
CARBON_CHECK(!buffer.has_errors()) << helper.DiagnoseErrors();
|
||||
}
|
||||
|
||||
state.counters["TokenRate"] = benchmark::Counter(
|
||||
NumTokens, benchmark::Counter::kIsIterationInvariantRate);
|
||||
}
|
||||
BENCHMARK(BM_ValidIdentifiers)
|
||||
// Benchmark a few ranges of identifier widths to cover different patterns
|
||||
// that emerge with small, medium, and longer identifiers.
|
||||
->Args({1, 3})
|
||||
->Args({3, 5})
|
||||
->Args({3, 16})
|
||||
->Args({12, 64});
|
||||
|
||||
} // namespace
|
||||
} // namespace Carbon::Testing
|
||||
Reference in New Issue
Block a user