Enhance the main lexer benchmark. (#3275)

This generalizes the main lexer benchmark's source generation to be
a bit more comprehensive, specifically including whitespace and
comments. Without these, we're missing a key part of the lexer's
performance.

This also tidies up a bit of the code and adds a more specific
distribution of the different factors in lexing based on analysis of
LLVM's source code. The enhancements to the source statistics script
that helped collect the data here will be in a separate PR.

With this, the output on my AMD cloud instance is:
```
-------------------------------------------------------------------------------------------------------------------------
Benchmark                                            Time             CPU   Iterations bytes_per_second tokens_per_second
-------------------------------------------------------------------------------------------------------------------------
BM_ValidKeywords                               2809011 ns      2808912 ns          243       212.616M/s         35.601M/s
BM_ValidIdentifiers<1, 64, false>             11461783 ns     11461652 ns           61       128.499M/s        8.72475M/s
BM_ValidIdentifiers<1, 1, true>                3397293 ns      3397240 ns          208       84.2158M/s        29.4357M/s
BM_ValidIdentifiers<3, 5, true>               14165143 ns     14164962 ns           51       40.3956M/s        7.05967M/s
BM_ValidIdentifiers<3, 16, true>              15283128 ns     15282583 ns           47       71.7623M/s         6.5434M/s
BM_ValidIdentifiers<12, 64, true>             17417323 ns     17417109 ns           41       219.007M/s        5.74148M/s
------------------------------------------------------------------------------------------------------------------------------------------
Benchmark                                            Time             CPU   Iterations bytes_per_second lines_per_second tokens_per_second
------------------------------------------------------------------------------------------------------------------------------------------
BM_RandomSource                                8132211 ns      8132227 ns           84       137.223M/s        3.9036M/s        12.2968M/s
BM_SpeedOfLightStrCpy                            29610 ns        29608 ns        24631       36.8065G/s       1072.17M/s        3.37745G/s
BM_SpeedOfLightDispatch<1>                     2144418 ns      2144421 ns          327       520.387M/s       14.8035M/s        46.6326M/s
BM_SpeedOfLightDispatch<2>                     1945954 ns      1945827 ns          351       573.498M/s       16.3144M/s         51.392M/s
BM_SpeedOfLightDispatch<4>                     2519565 ns      2519467 ns          292       442.923M/s       12.5999M/s        39.6909M/s
BM_SpeedOfLightDispatch<8>                     3011965 ns      3011968 ns          238       370.498M/s       10.5396M/s        33.2009M/s
BM_SpeedOfLightDispatch<16>                    4379575 ns      4379579 ns          160       254.803M/s       7.24841M/s        22.8332M/s
BM_SpeedOfLightDispatch<32>                    6678423 ns      6678353 ns          102       167.096M/s       4.75342M/s        14.9738M/s
BM_SpeedOfLightDispatch<MaxDispatchTargets>    9373075 ns      9372688 ns           75       119.062M/s       3.38697M/s        10.6693M/s
```

I've compared the profile of the `BM_RandomSource` benchmark with this
change and it largely corresponds to what I expect based on profiling
hand-crafted Carbon inputs. And as you can see, we're closing in on
lexing at least hitting the 10-million-lines-per-second mark. =]

---------

Co-authored-by: Richard Smith <richard@metafoo.co.uk>
This commit is contained in:
Chandler Carruth
2023-10-12 06:35:24 +00:00
committed by GitHub
co-authored by Richard Smith
parent 9c3e25decf
commit 2d735bbc51
+123 -26
View File
@@ -226,15 +226,70 @@ auto GetSymbolTokenTable() -> llvm::ArrayRef<TokenKind> {
return symbol_token_table_storage;
}
// Compute a random sequence of mixed symbols, keywords, and identifiers, with
// percentages of each according to the parameters.
auto RandomMixedSeq(int symbol_percent, int keyword_percent) -> std::string {
CARBON_CHECK(0 <= symbol_percent && symbol_percent <= 100)
<< "Must be a percent: [0, 100].";
CARBON_CHECK(0 <= keyword_percent && keyword_percent <= 100)
<< "Must be a percent: [0, 100].";
CARBON_CHECK((symbol_percent + keyword_percent) <= 100)
<< "Cannot have >100%.";
struct RandomSourceOptions {
int symbol_percent = 0;
int keyword_percent = 0;
int numeric_literal_percent = 0;
int string_literal_percent = 0;
int tokens_per_line = NumTokens;
int comment_line_percent = 0;
int blank_line_percent = 0;
void Validate() {
auto is_percentage = [](int n) { return 0 <= n && n <= 100; };
CARBON_CHECK(is_percentage(symbol_percent));
CARBON_CHECK(is_percentage(keyword_percent));
CARBON_CHECK(is_percentage(numeric_literal_percent));
CARBON_CHECK(is_percentage(string_literal_percent));
CARBON_CHECK(is_percentage(symbol_percent + keyword_percent +
numeric_literal_percent +
string_literal_percent));
CARBON_CHECK(tokens_per_line <= NumTokens);
CARBON_CHECK(NumTokens % tokens_per_line == 0)
<< "Tokens per line of " << tokens_per_line
<< " does not divide the number of tokens " << NumTokens;
CARBON_CHECK(is_percentage(comment_line_percent));
CARBON_CHECK(is_percentage(blank_line_percent));
// Ensure that comment and blank lines are less than 100% so we eventually
// produce a token line.
CARBON_CHECK(comment_line_percent + blank_line_percent < 100);
}
};
// Based on measurements of LLVM's source code, a rough approximation of the
// distribution of these kinds of tokens.
constexpr RandomSourceOptions DefaultSourceDist = {
.symbol_percent = 50,
.keyword_percent = 7,
.numeric_literal_percent = 17,
.string_literal_percent = 1,
// The median for LLVM is roughly 5.
.tokens_per_line = 5,
// Observed percentage of lines in LLVM.
.comment_line_percent = 22,
.blank_line_percent = 15,
};
// Compute random source code with a mixture of tokens and whitespace according
// to the options. The source isn't designed to be valid, or directly
// representative of real-world Carbon code. However, it tries to provide
// reasonable coverage of the different aspects of Carbon's lexer, such that for
// real world source code with distributions similar to the options provided the
// lexer performance will be roughly representative.
//
// TODO: Does not yet support generating numeric or string literals.
//
// TODO: The shape of lines is handled very arbitrarily and should vary more to
// avoid over-fitting to a specific shape (number of tokens, length of comment).
auto RandomSource(RandomSourceOptions options) -> std::string {
options.Validate();
static_assert((NumTokens % 100) == 0,
"The number of tokens must be divisible by 100 so that we can "
"easily scale integer percentages up to it.");
@@ -246,10 +301,10 @@ auto RandomMixedSeq(int symbol_percent, int keyword_percent) -> std::string {
// Build a list of StringRefs from the different types with the desired
// distribution, then shuffle that list.
std::array<llvm::StringRef, NumTokens> tokens;
llvm::OwningArrayRef<llvm::StringRef> tokens(NumTokens);
int num_symbols = (NumTokens / 100) * symbol_percent;
int num_keywords = (NumTokens / 100) * keyword_percent;
int num_symbols = (NumTokens / 100) * options.symbol_percent;
int num_keywords = (NumTokens / 100) * options.keyword_percent;
int num_identifiers = NumTokens - num_symbols - num_keywords;
CARBON_CHECK(num_identifiers == 0 || num_identifiers > 500)
<< "We require at least 500 identifiers as we need to collect a "
@@ -268,7 +323,48 @@ auto RandomMixedSeq(int symbol_percent, int keyword_percent) -> std::string {
}
std::shuffle(tokens.begin(), tokens.end(), absl::BitGen());
return llvm::join(tokens, " ");
// Distribute the tokens across lines as well as horizontal whitespace. The
// goal isn't to make any one line representative of anything, but to make the
// rough density of different kinds of whitespace roughly representative.
//
// TODO: This is a really coarse approach that just picks a fixed number of
// tokens per line rather than using some distribution with this as the median
// or mean.
llvm::SmallVector<std::string> lines;
// First place tokens onto each line.
for (auto i : llvm::seq(NumTokens / options.tokens_per_line)) {
lines.push_back("");
llvm::raw_string_ostream os(lines.back());
// Arbitrarily indent each line by two spaces.
os << " ";
llvm::ListSeparator sep(" ");
for (int j : llvm::seq(options.tokens_per_line)) {
os << sep << tokens[i * options.tokens_per_line + j];
}
}
// Next, synthesize blank and comment lines with the correct distribution.
int token_line_percent =
100 - options.blank_line_percent - options.comment_line_percent;
CARBON_CHECK(token_line_percent > 0);
int num_token_lines = lines.size();
int num_lines = num_token_lines * 100 / token_line_percent;
int num_blank_lines = num_lines * options.blank_line_percent / 100;
int num_comment_lines = num_lines - num_blank_lines - num_token_lines;
CARBON_CHECK(num_comment_lines >= 0);
lines.resize(num_lines);
for (auto& line :
llvm::MutableArrayRef(lines).slice(num_lines - num_comment_lines)) {
// TODO: We should vary the content and length, especially as the
// distribution is weirdly shaped with just over half the comment lines
// being blank and the median length of non-black comment lines being 64!
// This is a *very* coarse approximation of the mean at 30 characters long.
line = " // abcdefghijklmnopqrstuvwxyz";
}
// Now shuffle the lines.
std::shuffle(lines.begin(), lines.end(), absl::BitGen());
// And join them into the source string.
return llvm::join(lines, "\n");
}
class LexerBenchHelper {
@@ -354,10 +450,8 @@ BENCHMARK(BM_ValidIdentifiers<3, 5, /*Uniform=*/true>);
BENCHMARK(BM_ValidIdentifiers<3, 16, /*Uniform=*/true>);
BENCHMARK(BM_ValidIdentifiers<12, 64, /*Uniform=*/true>);
void BM_ValidMix(benchmark::State& state) {
int symbol_percent = state.range(0);
int keyword_percent = state.range(1);
std::string source = RandomMixedSeq(symbol_percent, keyword_percent);
void BM_RandomSource(benchmark::State& state) {
std::string source = RandomSource(DefaultSourceDist);
LexerBenchHelper helper(source);
for (auto _ : state) {
@@ -371,16 +465,15 @@ void BM_ValidMix(benchmark::State& state) {
state.SetBytesProcessed(state.iterations() * source.size());
state.counters["tokens_per_second"] = benchmark::Counter(
NumTokens, benchmark::Counter::kIsIterationInvariantRate);
state.counters["lines_per_second"] =
benchmark::Counter(llvm::StringRef(source).count('\n'),
benchmark::Counter::kIsIterationInvariantRate);
}
// The distributions between symbols, keywords, and identifiers here are
// guesses. Eventually, we should collect more data to help tune these, but
// hopefully the performance isn't too sensitive and we can just cover a wide
// range here.
BENCHMARK(BM_ValidMix)
->Args({10, 40})
->Args({25, 30})
->Args({50, 20})
->Args({75, 10});
BENCHMARK(BM_RandomSource);
// This is a speed-of-light benchmark that should reflect memory bandwidth
// (ideally) of simply reading all the source code. For speed-of-light we use
@@ -393,8 +486,7 @@ BENCHMARK(BM_ValidMix)
// to reflect whatever distribution is most realistic long-term. The
// bytes/second throughput is the important output of this routine.
auto BM_SpeedOfLightStrCpy(benchmark::State& state) -> void {
std::string source =
RandomMixedSeq(/*symbol_percent=*/25, /*keyword_percent=*/30);
std::string source = RandomSource(DefaultSourceDist);
// A buffer to write the null-terminated contents of `source` into.
llvm::OwningArrayRef<char> buffer(source.size() + 1);
@@ -409,6 +501,9 @@ auto BM_SpeedOfLightStrCpy(benchmark::State& state) -> void {
state.SetBytesProcessed(state.iterations() * source.size());
state.counters["tokens_per_second"] = benchmark::Counter(
NumTokens, benchmark::Counter::kIsIterationInvariantRate);
state.counters["lines_per_second"] =
benchmark::Counter(llvm::StringRef(source).count('\n'),
benchmark::Counter::kIsIterationInvariantRate);
}
BENCHMARK(BM_SpeedOfLightStrCpy);
@@ -528,8 +623,7 @@ constexpr DispatchTableT DispatchTable = []() {
template <int NumDispatchTargets>
auto BM_SpeedOfLightDispatch(benchmark::State& state) -> void {
std::string source =
RandomMixedSeq(/*symbol_percent=*/25, /*keyword_percent=*/30);
std::string source = RandomSource(DefaultSourceDist);
// A buffer to write to, simulating some minimal write traffic.
llvm::OwningArrayRef<char> buffer(source.size());
@@ -551,6 +645,9 @@ auto BM_SpeedOfLightDispatch(benchmark::State& state) -> void {
state.SetBytesProcessed(state.iterations() * source.size());
state.counters["tokens_per_second"] = benchmark::Counter(
NumTokens, benchmark::Counter::kIsIterationInvariantRate);
state.counters["lines_per_second"] =
benchmark::Counter(llvm::StringRef(source).count('\n'),
benchmark::Counter::kIsIterationInvariantRate);
}
BENCHMARK(BM_SpeedOfLightDispatch<1>);
BENCHMARK(BM_SpeedOfLightDispatch<2>);