mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-10-05 20:01:14 +01:00
Enhance the main lexer benchmark. (#3275)
This generalizes the main lexer benchmark's source generation to be a bit more comprehensive, specifically including whitespace and comments. Without these, we're missing a key part of the lexer's performance. This also tidies up a bit of the code and adds a more specific distribution of the different factors in lexing based on analysis of LLVM's source code. The enhancements to the source statistics script that helped collect the data here will be in a separate PR. With this, the output on my AMD cloud instance is: ``` ------------------------------------------------------------------------------------------------------------------------- Benchmark Time CPU Iterations bytes_per_second tokens_per_second ------------------------------------------------------------------------------------------------------------------------- BM_ValidKeywords 2809011 ns 2808912 ns 243 212.616M/s 35.601M/s BM_ValidIdentifiers<1, 64, false> 11461783 ns 11461652 ns 61 128.499M/s 8.72475M/s BM_ValidIdentifiers<1, 1, true> 3397293 ns 3397240 ns 208 84.2158M/s 29.4357M/s BM_ValidIdentifiers<3, 5, true> 14165143 ns 14164962 ns 51 40.3956M/s 7.05967M/s BM_ValidIdentifiers<3, 16, true> 15283128 ns 15282583 ns 47 71.7623M/s 6.5434M/s BM_ValidIdentifiers<12, 64, true> 17417323 ns 17417109 ns 41 219.007M/s 5.74148M/s ------------------------------------------------------------------------------------------------------------------------------------------ Benchmark Time CPU Iterations bytes_per_second lines_per_second tokens_per_second ------------------------------------------------------------------------------------------------------------------------------------------ BM_RandomSource 8132211 ns 8132227 ns 84 137.223M/s 3.9036M/s 12.2968M/s BM_SpeedOfLightStrCpy 29610 ns 29608 ns 24631 36.8065G/s 1072.17M/s 3.37745G/s BM_SpeedOfLightDispatch<1> 2144418 ns 2144421 ns 327 520.387M/s 14.8035M/s 46.6326M/s BM_SpeedOfLightDispatch<2> 1945954 ns 1945827 ns 351 573.498M/s 16.3144M/s 51.392M/s BM_SpeedOfLightDispatch<4> 2519565 ns 2519467 ns 292 442.923M/s 12.5999M/s 39.6909M/s BM_SpeedOfLightDispatch<8> 3011965 ns 3011968 ns 238 370.498M/s 10.5396M/s 33.2009M/s BM_SpeedOfLightDispatch<16> 4379575 ns 4379579 ns 160 254.803M/s 7.24841M/s 22.8332M/s BM_SpeedOfLightDispatch<32> 6678423 ns 6678353 ns 102 167.096M/s 4.75342M/s 14.9738M/s BM_SpeedOfLightDispatch<MaxDispatchTargets> 9373075 ns 9372688 ns 75 119.062M/s 3.38697M/s 10.6693M/s ``` I've compared the profile of the `BM_RandomSource` benchmark with this change and it largely corresponds to what I expect based on profiling hand-crafted Carbon inputs. And as you can see, we're closing in on lexing at least hitting the 10-million-lines-per-second mark. =] --------- Co-authored-by: Richard Smith <richard@metafoo.co.uk>
This commit is contained in:
co-authored by
Richard Smith
parent
9c3e25decf
commit
2d735bbc51
@@ -226,15 +226,70 @@ auto GetSymbolTokenTable() -> llvm::ArrayRef<TokenKind> {
|
||||
return symbol_token_table_storage;
|
||||
}
|
||||
|
||||
// Compute a random sequence of mixed symbols, keywords, and identifiers, with
|
||||
// percentages of each according to the parameters.
|
||||
auto RandomMixedSeq(int symbol_percent, int keyword_percent) -> std::string {
|
||||
CARBON_CHECK(0 <= symbol_percent && symbol_percent <= 100)
|
||||
<< "Must be a percent: [0, 100].";
|
||||
CARBON_CHECK(0 <= keyword_percent && keyword_percent <= 100)
|
||||
<< "Must be a percent: [0, 100].";
|
||||
CARBON_CHECK((symbol_percent + keyword_percent) <= 100)
|
||||
<< "Cannot have >100%.";
|
||||
struct RandomSourceOptions {
|
||||
int symbol_percent = 0;
|
||||
int keyword_percent = 0;
|
||||
int numeric_literal_percent = 0;
|
||||
int string_literal_percent = 0;
|
||||
|
||||
int tokens_per_line = NumTokens;
|
||||
|
||||
int comment_line_percent = 0;
|
||||
int blank_line_percent = 0;
|
||||
|
||||
void Validate() {
|
||||
auto is_percentage = [](int n) { return 0 <= n && n <= 100; };
|
||||
CARBON_CHECK(is_percentage(symbol_percent));
|
||||
CARBON_CHECK(is_percentage(keyword_percent));
|
||||
CARBON_CHECK(is_percentage(numeric_literal_percent));
|
||||
CARBON_CHECK(is_percentage(string_literal_percent));
|
||||
CARBON_CHECK(is_percentage(symbol_percent + keyword_percent +
|
||||
numeric_literal_percent +
|
||||
string_literal_percent));
|
||||
|
||||
CARBON_CHECK(tokens_per_line <= NumTokens);
|
||||
CARBON_CHECK(NumTokens % tokens_per_line == 0)
|
||||
<< "Tokens per line of " << tokens_per_line
|
||||
<< " does not divide the number of tokens " << NumTokens;
|
||||
|
||||
CARBON_CHECK(is_percentage(comment_line_percent));
|
||||
CARBON_CHECK(is_percentage(blank_line_percent));
|
||||
|
||||
// Ensure that comment and blank lines are less than 100% so we eventually
|
||||
// produce a token line.
|
||||
CARBON_CHECK(comment_line_percent + blank_line_percent < 100);
|
||||
}
|
||||
};
|
||||
|
||||
// Based on measurements of LLVM's source code, a rough approximation of the
|
||||
// distribution of these kinds of tokens.
|
||||
constexpr RandomSourceOptions DefaultSourceDist = {
|
||||
.symbol_percent = 50,
|
||||
.keyword_percent = 7,
|
||||
.numeric_literal_percent = 17,
|
||||
.string_literal_percent = 1,
|
||||
|
||||
// The median for LLVM is roughly 5.
|
||||
.tokens_per_line = 5,
|
||||
|
||||
// Observed percentage of lines in LLVM.
|
||||
.comment_line_percent = 22,
|
||||
.blank_line_percent = 15,
|
||||
};
|
||||
|
||||
// Compute random source code with a mixture of tokens and whitespace according
|
||||
// to the options. The source isn't designed to be valid, or directly
|
||||
// representative of real-world Carbon code. However, it tries to provide
|
||||
// reasonable coverage of the different aspects of Carbon's lexer, such that for
|
||||
// real world source code with distributions similar to the options provided the
|
||||
// lexer performance will be roughly representative.
|
||||
//
|
||||
// TODO: Does not yet support generating numeric or string literals.
|
||||
//
|
||||
// TODO: The shape of lines is handled very arbitrarily and should vary more to
|
||||
// avoid over-fitting to a specific shape (number of tokens, length of comment).
|
||||
auto RandomSource(RandomSourceOptions options) -> std::string {
|
||||
options.Validate();
|
||||
static_assert((NumTokens % 100) == 0,
|
||||
"The number of tokens must be divisible by 100 so that we can "
|
||||
"easily scale integer percentages up to it.");
|
||||
@@ -246,10 +301,10 @@ auto RandomMixedSeq(int symbol_percent, int keyword_percent) -> std::string {
|
||||
|
||||
// Build a list of StringRefs from the different types with the desired
|
||||
// distribution, then shuffle that list.
|
||||
std::array<llvm::StringRef, NumTokens> tokens;
|
||||
llvm::OwningArrayRef<llvm::StringRef> tokens(NumTokens);
|
||||
|
||||
int num_symbols = (NumTokens / 100) * symbol_percent;
|
||||
int num_keywords = (NumTokens / 100) * keyword_percent;
|
||||
int num_symbols = (NumTokens / 100) * options.symbol_percent;
|
||||
int num_keywords = (NumTokens / 100) * options.keyword_percent;
|
||||
int num_identifiers = NumTokens - num_symbols - num_keywords;
|
||||
CARBON_CHECK(num_identifiers == 0 || num_identifiers > 500)
|
||||
<< "We require at least 500 identifiers as we need to collect a "
|
||||
@@ -268,7 +323,48 @@ auto RandomMixedSeq(int symbol_percent, int keyword_percent) -> std::string {
|
||||
}
|
||||
std::shuffle(tokens.begin(), tokens.end(), absl::BitGen());
|
||||
|
||||
return llvm::join(tokens, " ");
|
||||
// Distribute the tokens across lines as well as horizontal whitespace. The
|
||||
// goal isn't to make any one line representative of anything, but to make the
|
||||
// rough density of different kinds of whitespace roughly representative.
|
||||
//
|
||||
// TODO: This is a really coarse approach that just picks a fixed number of
|
||||
// tokens per line rather than using some distribution with this as the median
|
||||
// or mean.
|
||||
llvm::SmallVector<std::string> lines;
|
||||
// First place tokens onto each line.
|
||||
for (auto i : llvm::seq(NumTokens / options.tokens_per_line)) {
|
||||
lines.push_back("");
|
||||
llvm::raw_string_ostream os(lines.back());
|
||||
// Arbitrarily indent each line by two spaces.
|
||||
os << " ";
|
||||
llvm::ListSeparator sep(" ");
|
||||
for (int j : llvm::seq(options.tokens_per_line)) {
|
||||
os << sep << tokens[i * options.tokens_per_line + j];
|
||||
}
|
||||
}
|
||||
|
||||
// Next, synthesize blank and comment lines with the correct distribution.
|
||||
int token_line_percent =
|
||||
100 - options.blank_line_percent - options.comment_line_percent;
|
||||
CARBON_CHECK(token_line_percent > 0);
|
||||
int num_token_lines = lines.size();
|
||||
int num_lines = num_token_lines * 100 / token_line_percent;
|
||||
int num_blank_lines = num_lines * options.blank_line_percent / 100;
|
||||
int num_comment_lines = num_lines - num_blank_lines - num_token_lines;
|
||||
CARBON_CHECK(num_comment_lines >= 0);
|
||||
lines.resize(num_lines);
|
||||
for (auto& line :
|
||||
llvm::MutableArrayRef(lines).slice(num_lines - num_comment_lines)) {
|
||||
// TODO: We should vary the content and length, especially as the
|
||||
// distribution is weirdly shaped with just over half the comment lines
|
||||
// being blank and the median length of non-black comment lines being 64!
|
||||
// This is a *very* coarse approximation of the mean at 30 characters long.
|
||||
line = " // abcdefghijklmnopqrstuvwxyz";
|
||||
}
|
||||
// Now shuffle the lines.
|
||||
std::shuffle(lines.begin(), lines.end(), absl::BitGen());
|
||||
// And join them into the source string.
|
||||
return llvm::join(lines, "\n");
|
||||
}
|
||||
|
||||
class LexerBenchHelper {
|
||||
@@ -354,10 +450,8 @@ BENCHMARK(BM_ValidIdentifiers<3, 5, /*Uniform=*/true>);
|
||||
BENCHMARK(BM_ValidIdentifiers<3, 16, /*Uniform=*/true>);
|
||||
BENCHMARK(BM_ValidIdentifiers<12, 64, /*Uniform=*/true>);
|
||||
|
||||
void BM_ValidMix(benchmark::State& state) {
|
||||
int symbol_percent = state.range(0);
|
||||
int keyword_percent = state.range(1);
|
||||
std::string source = RandomMixedSeq(symbol_percent, keyword_percent);
|
||||
void BM_RandomSource(benchmark::State& state) {
|
||||
std::string source = RandomSource(DefaultSourceDist);
|
||||
|
||||
LexerBenchHelper helper(source);
|
||||
for (auto _ : state) {
|
||||
@@ -371,16 +465,15 @@ void BM_ValidMix(benchmark::State& state) {
|
||||
state.SetBytesProcessed(state.iterations() * source.size());
|
||||
state.counters["tokens_per_second"] = benchmark::Counter(
|
||||
NumTokens, benchmark::Counter::kIsIterationInvariantRate);
|
||||
state.counters["lines_per_second"] =
|
||||
benchmark::Counter(llvm::StringRef(source).count('\n'),
|
||||
benchmark::Counter::kIsIterationInvariantRate);
|
||||
}
|
||||
// The distributions between symbols, keywords, and identifiers here are
|
||||
// guesses. Eventually, we should collect more data to help tune these, but
|
||||
// hopefully the performance isn't too sensitive and we can just cover a wide
|
||||
// range here.
|
||||
BENCHMARK(BM_ValidMix)
|
||||
->Args({10, 40})
|
||||
->Args({25, 30})
|
||||
->Args({50, 20})
|
||||
->Args({75, 10});
|
||||
BENCHMARK(BM_RandomSource);
|
||||
|
||||
// This is a speed-of-light benchmark that should reflect memory bandwidth
|
||||
// (ideally) of simply reading all the source code. For speed-of-light we use
|
||||
@@ -393,8 +486,7 @@ BENCHMARK(BM_ValidMix)
|
||||
// to reflect whatever distribution is most realistic long-term. The
|
||||
// bytes/second throughput is the important output of this routine.
|
||||
auto BM_SpeedOfLightStrCpy(benchmark::State& state) -> void {
|
||||
std::string source =
|
||||
RandomMixedSeq(/*symbol_percent=*/25, /*keyword_percent=*/30);
|
||||
std::string source = RandomSource(DefaultSourceDist);
|
||||
|
||||
// A buffer to write the null-terminated contents of `source` into.
|
||||
llvm::OwningArrayRef<char> buffer(source.size() + 1);
|
||||
@@ -409,6 +501,9 @@ auto BM_SpeedOfLightStrCpy(benchmark::State& state) -> void {
|
||||
state.SetBytesProcessed(state.iterations() * source.size());
|
||||
state.counters["tokens_per_second"] = benchmark::Counter(
|
||||
NumTokens, benchmark::Counter::kIsIterationInvariantRate);
|
||||
state.counters["lines_per_second"] =
|
||||
benchmark::Counter(llvm::StringRef(source).count('\n'),
|
||||
benchmark::Counter::kIsIterationInvariantRate);
|
||||
}
|
||||
BENCHMARK(BM_SpeedOfLightStrCpy);
|
||||
|
||||
@@ -528,8 +623,7 @@ constexpr DispatchTableT DispatchTable = []() {
|
||||
|
||||
template <int NumDispatchTargets>
|
||||
auto BM_SpeedOfLightDispatch(benchmark::State& state) -> void {
|
||||
std::string source =
|
||||
RandomMixedSeq(/*symbol_percent=*/25, /*keyword_percent=*/30);
|
||||
std::string source = RandomSource(DefaultSourceDist);
|
||||
|
||||
// A buffer to write to, simulating some minimal write traffic.
|
||||
llvm::OwningArrayRef<char> buffer(source.size());
|
||||
@@ -551,6 +645,9 @@ auto BM_SpeedOfLightDispatch(benchmark::State& state) -> void {
|
||||
state.SetBytesProcessed(state.iterations() * source.size());
|
||||
state.counters["tokens_per_second"] = benchmark::Counter(
|
||||
NumTokens, benchmark::Counter::kIsIterationInvariantRate);
|
||||
state.counters["lines_per_second"] =
|
||||
benchmark::Counter(llvm::StringRef(source).count('\n'),
|
||||
benchmark::Counter::kIsIterationInvariantRate);
|
||||
}
|
||||
BENCHMARK(BM_SpeedOfLightDispatch<1>);
|
||||
BENCHMARK(BM_SpeedOfLightDispatch<2>);
|
||||
|
||||
Reference in New Issue
Block a user