mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-09-24 20:50:13 +01:00
Replaces the callback-based `ForEach` methods on `RawHashtable`, `Map`,
and
`Set` with a range object supporting range-for loops, structured
bindings, and
the standard range concepts.
- Adds `.entries()` on `Map`, `Set`, and `RawHashtable`, returning a
range that
models `std::ranges::forward_range` and `std::ranges::common_range`.
Obtaining one is an explicit call rather than `begin()`/`end()` on the
container, as scanning a whole table is costly and shouldn't be hidden.
- Iterating a `Map` yields a `std::pair` of key and value references,
which
fits in two registers and is returned without being materialized in
memory.
- `Map::Range` and `Set::Range` are aliases of the raw hashtable's range
rather
than wrappers around it. The raw iterator produces the user-facing
reference
itself -- a `KeyT&` for a set, a pair of references for a map -- picked
by
`StorageEntry`, which is already specialized on whether there is a value
type. That leaves one iterator to reason about instead of three.
- Deletes the rvalue `.entries()` overloads on the owning containers, as
a
range built from a temporary table would dangle. Views don't own their
storage, so the operation remains available on them.
- In release builds, the walk over the groups is a single induction
variable: a
negative byte offset counting up to zero, anchored at the ends of the
metadata and entry arrays. Both arrays are then reached by indexed
addressing
off a base that stays put, and the entry pointer is formed only once a
group
with a present entry has been found.
- In debug builds, the range hashes the table's metadata when it is
built and
re-checks that hash when it is destroyed, catching mutation of the table
while a range is live. It also picks a random starting group and a
random odd
group stride, which varies the traversal order between ranges while
still
visiting every group exactly once. That entropy is drawn when the range
is
built rather than in `begin()`, so `begin()` stays a pure function of
the
range and the multi-pass guarantee holds.
- Removes `ForEachEntry` and all of its callers.
Measured against the iteration benchmark added in its own commit, a
traversal is at or ahead of what the callback compiled to across nearly
the
whole size range. The largest tables spend 3-5% fewer cycles, small
`Set`s as
much as 24% fewer, and instruction counts stay within about 1%. What
remains
behind is a handful of mid-sized `Map`s by up to 1%, and `Set` at 65536,
which
sits at exactly half its load factor, by 2%.
Both revisions were built with `-c opt --copt=-march=x86-64-v3` and
compared
with:
```
./scripts/bench_runner.py --exp_benchmark=... --base_benchmark=... \
--benchmark_args=--benchmark_perf_counters=INSTRUCTIONS,CYCLES \
--benchmark_args='--benchmark_filter=(Set|Map)Iterate<(Set|Map)<' \
--extra_metrics_filter='(INSTRUCTIONS|CYCLES)'
```
Trimmed below to the primary integer configurations and to the two
counters;
the pointer- and string-keyed configurations follow the same pattern.
```
Benchmark ┃ CYCLES ┃ INSTRUCTIONS
━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
BM_MapIterate<Map<int, int>>/1....... │ 👍 -6.032% p=1.14e-05 │ ?? p=0.752
baseline: │ 12.06 ± 1.520% │ 64 ± 3.125%
experiment: │ 11.33 ± 2.765% │ 65.5 ± 3.817%
│ │
BM_MapIterate<Map<int, int>>/2....... │ ?? p=0.155 │ ?? p=0.343
baseline: │ 7.587 ± 1.285% │ 41 ± 0.000%
experiment: │ 7.652 ± 0.865% │ 41 ± 2.439%
│ │
BM_MapIterate<Map<int, int>>/3....... │ ?? p=0.343 │ 👍 -1.020% p=0.0039
baseline: │ 6.663 ± 4.260% │ 32.67 ± 2.041%
experiment: │ 6.368 ± 12.224% │ 32.33 ± 2.062%
│ │
BM_MapIterate<Map<int, int>>/4....... │ ?? p=0.343 │ 👍 -1.786% p=0.0297
baseline: │ 6.091 ± 15.470% │ 28 ± 3.571%
experiment: │ 5.957 ± 8.932% │ 27.5 ± 3.636%
│ │
BM_MapIterate<Map<int, int>>/8....... │ ?? p=0.323 │ 👍 -1.220% p=0.000148
baseline: │ 4.845 ± 0.800% │ 20.5 ± 0.000%
experiment: │ 4.814 ± 3.585% │ 20.25 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/16...... │ 👍 -2.195% p=0.00908 │ 👍 0.769% p=6.58e-06
baseline: │ 4.312 ± 0.187% │ 16.25 ± 0.000%
experiment: │ 4.218 ± 2.368% │ 16.13 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/32...... │ ?? p=0.236 │ 👍 0.442% p=9.53e-06
baseline: │ 4.051 ± 1.084% │ 14.13 ± 0.000%
experiment: │ 4.063 ± 0.737% │ 14.06 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/64...... │ ?? p=0.693 │ 👎 0.227% p=4.52e-06
baseline: │ 4.021 ± 0.239% │ 13.75 ± 0.000%
experiment: │ 4.019 ± 0.417% │ 13.78 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/256..... │ 👍 0.360% p=0.00119 │ 👎 0.754% p=1.37e-05
baseline: │ 3.996 ± 0.173% │ 13.47 ± 0.000%
experiment: │ 3.982 ± 0.272% │ 13.57 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/4096.... │ 👍 0.581% p=1.96e-05 │ 👎 0.923% p=1.96e-05
baseline: │ 4.005 ± 0.816% │ 13.38 ± 0.000%
experiment: │ 3.981 ± 0.192% │ 13.5 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/65536... │ 👍 -4.957% p=1.14e-05 │ 👎 0.934% p=1.14e-05
baseline: │ 5.307 ± 0.501% │ 13.38 ± 0.000%
experiment: │ 5.044 ± 1.746% │ 13.5 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/1048576. │ 👍 -3.947% p=9.09e-05 │ 👎 0.935% p=3.3e-05
baseline: │ 6.074 ± 0.807% │ 13.38 ± 0.000%
experiment: │ 5.834 ± 2.159% │ 13.5 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/16777216 │ ?? p=0.155 │ 👎 0.935% p=2.11e-05
baseline: │ 5.082 ± 3.650% │ 13.38 ± 0.000%
experiment: │ 5.012 ± 1.316% │ 13.5 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/56...... │ 👎 0.825% p=0.0268 │ 👍 0.270% p=1.14e-05
baseline: │ 3.918 ± 0.501% │ 13.21 ± 0.000%
experiment: │ 3.951 ± 0.342% │ 13.18 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/224..... │ 👎 0.788% p=0.000504 │ 👎 0.346% p=1.64e-05
baseline: │ 3.895 ± 0.111% │ 12.89 ± 0.000%
experiment: │ 3.926 ± 0.285% │ 12.94 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/3584.... │ 👎 1.028% p=0.000148 │ 👎 0.545% p=1.14e-05
baseline: │ 3.913 ± 0.427% │ 12.79 ± 0.000%
experiment: │ 3.954 ± 0.325% │ 12.86 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/57344... │ ?? p=0.236 │ 👎 0.558% p=2.55e-06
baseline: │ 4.574 ± 0.721% │ 12.79 ± 0.000%
experiment: │ 4.51 ± 3.709% │ 12.86 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/917504.. │ 👍 -3.826% p=6.58e-06 │ 👎 0.559% p=2.33e-05
baseline: │ 5.221 ± 0.507% │ 12.79 ± 0.000%
experiment: │ 5.021 ± 0.556% │ 12.86 ± 0.000%
│ │
BM_MapIterate<Map<int, int>>/14680064 │ 👍 -3.839% p=1.37e-05 │ 👎 0.559% p=3.31e-05
baseline: │ 5.129 ± 1.194% │ 12.79 ± 0.000%
experiment: │ 4.932 ± 1.475% │ 12.86 ± 0.000%
│ │
Benchmark ┃ CYCLES ┃ INSTRUCTIONS
━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━━━━━━━━━━━━━
BM_SetIterate<Set<int>>/1....... │ 👍 -3.104% p=0.000583 │ ?? p=0.206
baseline: │ 11.2 ± 6.323% │ 60 ± 3.333%
experiment: │ 10.85 ± 5.820% │ 61 ± 3.279%
│ │
BM_SetIterate<Set<int>>/2....... │ 👍 -7.037% p=0.0362 │ ?? p=0.155
baseline: │ 7.086 ± 16.857% │ 37 ± 0.000%
experiment: │ 6.587 ± 0.479% │ 36 ± 4.167%
│ │
BM_SetIterate<Set<int>>/3....... │ 👎 1.400% p=2.34e-05 │ 👍 -1.163% p=0.00136
baseline: │ 5.363 ± 0.463% │ 28.67 ± 2.326%
experiment: │ 5.438 ± 32.763% │ 28.33 ± 1.176%
│ │
BM_SetIterate<Set<int>>/4....... │ ?? p=0.968 │ ?? p=0.286
baseline: │ 4.642 ± 32.751% │ 23.5 ± 2.128%
experiment: │ 4.658 ± 38.416% │ 23.63 ± 3.704%
│ │
BM_SetIterate<Set<int>>/8....... │ 👍 -23.823% p=3.74e-06 │ 👍 -1.515% p=5.52e-05
baseline: │ 4.701 ± 6.589% │ 16.5 ± 0.000%
experiment: │ 3.581 ± 7.790% │ 16.25 ± 0.000%
│ │
BM_SetIterate<Set<int>>/16...... │ 👍 -4.502% p=1.37e-05 │ 👍 -1.020% p=3.31e-05
baseline: │ 3.124 ± 0.585% │ 12.25 ± 0.000%
experiment: │ 2.983 ± 0.625% │ 12.13 ± 0.000%
│ │
BM_SetIterate<Set<int>>/32...... │ 👍 -4.032% p=5.46e-06 │ 👍 0.617% p=1.96e-05
baseline: │ 2.957 ± 0.260% │ 10.13 ± 0.000%
experiment: │ 2.838 ± 0.434% │ 10.06 ± 0.000%
│ │
BM_SetIterate<Set<int>>/64...... │ 👍 -5.054% p=4.52e-06 │ 👎 0.321% p=1.37e-05
baseline: │ 2.937 ± 0.301% │ 9.75 ± 0.000%
experiment: │ 2.788 ± 1.143% │ 9.781 ± 0.000%
│ │
BM_SetIterate<Set<int>>/256..... │ 👍 -5.325% p=1.14e-05 │ 👎 1.073% p=6.58e-06
baseline: │ 2.916 ± 0.220% │ 9.469 ± 0.000%
experiment: │ 2.761 ± 0.142% │ 9.57 ± 0.000%
│ │
BM_SetIterate<Set<int>>/4096.... │ 👍 -4.865% p=4.52e-06 │ 👎 1.317% p=2.34e-05
baseline: │ 2.921 ± 0.194% │ 9.381 ± 0.000%
experiment: │ 2.779 ± 0.224% │ 9.504 ± 0.000%
│ │
BM_SetIterate<Set<int>>/65536... │ 👎 1.961% p=3.93e-05 │ 👎 1.332% p=1.49e-05
baseline: │ 4.015 ± 0.482% │ 9.375 ± 0.000%
experiment: │ 4.094 ± 0.613% │ 9.5 ± 0.000%
│ │
BM_SetIterate<Set<int>>/1048576. │ 👍 -4.843% p=1.14e-05 │ 👎 1.333% p=5.38e-06
baseline: │ 5.239 ± 0.144% │ 9.375 ± 0.000%
experiment: │ 4.986 ± 0.139% │ 9.5 ± 0.000%
│ │
BM_SetIterate<Set<int>>/16777216 │ 👍 0.840% p=0.0362 │ 👎 1.333% p=2.52e-06
baseline: │ 3.719 ± 1.420% │ 9.375 ± 0.000%
experiment: │ 3.688 ± 1.308% │ 9.5 ± 0.000%
│ │
BM_SetIterate<Set<int>>/56...... │ 👍 -2.857% p=9.53e-06 │ 👍 0.388% p=3.31e-05
baseline: │ 2.942 ± 0.439% │ 9.214 ± 0.000%
experiment: │ 2.858 ± 0.619% │ 9.179 ± 0.000%
│ │
BM_SetIterate<Set<int>>/224..... │ 👍 -2.161% p=2.34e-05 │ 👎 0.502% p=4.52e-06
baseline: │ 2.888 ± 0.347% │ 8.893 ± 0.000%
experiment: │ 2.826 ± 0.450% │ 8.938 ± 0.000%
│ │
BM_SetIterate<Set<int>>/3584.... │ 👍 -1.750% p=6.58e-06 │ 👎 0.793% p=2.34e-05
baseline: │ 2.89 ± 0.261% │ 8.792 ± 0.000%
experiment: │ 2.84 ± 0.411% │ 8.862 ± 0.000%
│ │
BM_SetIterate<Set<int>>/57344... │ ?? p=0.502 │ 👎 0.812% p=2.78e-05
baseline: │ 3.684 ± 4.246% │ 8.786 ± 0.000%
experiment: │ 3.644 ± 4.431% │ 8.857 ± 0.000%
│ │
BM_SetIterate<Set<int>>/917504.. │ 👍 -2.629% p=0.000148 │ 👎 0.813% p=3.08e-06
baseline: │ 4.372 ± 0.693% │ 8.786 ± 0.000%
experiment: │ 4.257 ± 0.210% │ 8.857 ± 0.000%
│ │
BM_SetIterate<Set<int>>/14680064 │ 👍 -2.927% p=0.0219 │ 👎 0.813% p=3.03e-06
baseline: │ 4.154 ± 3.286% │ 8.786 ± 0.000%
experiment: │ 4.032 ± 3.198% │ 8.857 ± 0.000%
│ │
```
Assisted-by: Antigravity with Opus
449 lines
16 KiB
C++
449 lines
16 KiB
C++
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
|
|
// Exceptions. See /LICENSE for license information.
|
|
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
|
|
|
#include <benchmark/benchmark.h>
|
|
|
|
#include <type_traits>
|
|
|
|
#include "absl/container/flat_hash_set.h"
|
|
#include "common/raw_hashtable_benchmark_helpers.h"
|
|
#include "common/set.h"
|
|
#include "llvm/ADT/DenseSet.h"
|
|
|
|
namespace Carbon {
|
|
namespace {
|
|
|
|
using RawHashtable::CarbonHashDI;
|
|
using RawHashtable::GetKeysAndHitKeys;
|
|
using RawHashtable::GetKeysAndMissKeys;
|
|
using RawHashtable::HitArgs;
|
|
using RawHashtable::ReportTableMetrics;
|
|
using RawHashtable::SizeArgs;
|
|
using RawHashtable::ValueToBool;
|
|
|
|
template <typename SetT>
|
|
struct IsCarbonSetImpl : std::false_type {};
|
|
template <typename KT, int MinSmallSize>
|
|
struct IsCarbonSetImpl<Set<KT, MinSmallSize>> : std::true_type {};
|
|
|
|
template <typename SetT>
|
|
static constexpr bool IsCarbonSet = IsCarbonSetImpl<SetT>::value;
|
|
|
|
// A wrapper around various set types that we specialize to implement a common
|
|
// API used in the benchmarks for various different map data structures that
|
|
// support different APIs. The primary template assumes a roughly
|
|
// `std::unordered_set` API design, and types with a different API design are
|
|
// supported through specializations.
|
|
template <typename InSetT>
|
|
struct SetWrapperImpl {
|
|
using SetT = InSetT;
|
|
using KeyT = SetT::key_type;
|
|
|
|
SetT s;
|
|
|
|
auto BenchContains(KeyT k) -> bool { return s.find(k) != s.end(); }
|
|
|
|
auto BenchLookup(KeyT k) -> bool {
|
|
auto it = s.find(k);
|
|
if (it == s.end()) {
|
|
return false;
|
|
}
|
|
// We expect keys to always convert to `true` so directly return that here.
|
|
return ValueToBool(*it);
|
|
}
|
|
|
|
auto BenchInsert(KeyT k) -> bool {
|
|
auto result = s.insert(k);
|
|
return result.second;
|
|
}
|
|
|
|
auto BenchErase(KeyT k) -> bool { return s.erase(k) != 0; }
|
|
|
|
// Visits every key in the set, calling `cb` with each one. Each set type is
|
|
// expected to traverse using whatever API it provides for this, so that the
|
|
// benchmark measures iterating the set rather than any specific iteration
|
|
// API.
|
|
template <typename CallbackT>
|
|
auto BenchIterate(CallbackT cb) -> void {
|
|
for (const auto& k : s) {
|
|
cb(k);
|
|
}
|
|
}
|
|
};
|
|
|
|
// Explicit (partial) specialization for the Carbon map type that uses its
|
|
// different API design.
|
|
template <typename KT, int MinSmallSize>
|
|
struct SetWrapperImpl<Set<KT, MinSmallSize>> {
|
|
using SetT = Set<KT, MinSmallSize>;
|
|
using KeyT = KT;
|
|
|
|
SetT s;
|
|
|
|
auto BenchContains(KeyT k) -> bool { return s.Contains(k); }
|
|
|
|
auto BenchLookup(KeyT k) -> bool {
|
|
auto result = s.Lookup(k);
|
|
if (!result) {
|
|
return false;
|
|
}
|
|
return ValueToBool(result.key());
|
|
}
|
|
|
|
auto BenchInsert(KeyT k) -> bool {
|
|
auto result = s.Insert(k);
|
|
return result.is_inserted();
|
|
}
|
|
|
|
auto BenchErase(KeyT k) -> bool { return s.Erase(k); }
|
|
|
|
template <typename CallbackT>
|
|
auto BenchIterate(CallbackT cb) -> void {
|
|
for (const auto& k : s.entries()) {
|
|
cb(k);
|
|
}
|
|
}
|
|
};
|
|
|
|
// Provide a way to override the Carbon Set specific benchmark runs with another
|
|
// hashtable implementation. When building, you can use one of these enum names
|
|
// in a macro define such as `-DCARBON_SET_BENCH_OVERRIDE=Name` in order to
|
|
// trigger a specific override for the `Set` type benchmarks. This is used to
|
|
// get before/after runs that compare the performance of Carbon's Set versus
|
|
// other implementations.
|
|
enum class SetOverride : uint8_t {
|
|
Abseil,
|
|
LLVM,
|
|
LLVMAndCarbonHash,
|
|
};
|
|
template <typename SetT, SetOverride Override>
|
|
struct SetWrapperOverride : SetWrapperImpl<SetT> {};
|
|
|
|
template <typename KeyT, int MinSmallSize>
|
|
struct SetWrapperOverride<Set<KeyT, MinSmallSize>, SetOverride::Abseil>
|
|
: SetWrapperImpl<absl::flat_hash_set<KeyT>> {};
|
|
|
|
template <typename KeyT, int MinSmallSize>
|
|
struct SetWrapperOverride<Set<KeyT, MinSmallSize>, SetOverride::LLVM>
|
|
: SetWrapperImpl<llvm::DenseSet<KeyT>> {};
|
|
|
|
template <typename KeyT, int MinSmallSize>
|
|
struct SetWrapperOverride<Set<KeyT, MinSmallSize>,
|
|
SetOverride::LLVMAndCarbonHash>
|
|
: SetWrapperImpl<llvm::DenseSet<KeyT, CarbonHashDI<KeyT>>> {};
|
|
|
|
#ifndef CARBON_SET_BENCH_OVERRIDE
|
|
template <typename SetT>
|
|
using SetWrapper = SetWrapperImpl<SetT>;
|
|
#else
|
|
template <typename SetT>
|
|
using SetWrapper =
|
|
SetWrapperOverride<SetT, SetOverride::CARBON_SET_BENCH_OVERRIDE>;
|
|
#endif
|
|
|
|
// Reports extra statistics about the table, when it is in fact a Carbon table.
|
|
// Note that this has to inspect the *wrapped* type in order to work correctly
|
|
// when the Carbon benchmarks are overridden with another implementation.
|
|
template <typename SetT>
|
|
auto ReportMetrics(const SetWrapper<SetT>& s_wrapper, benchmark::State& state)
|
|
-> void {
|
|
if constexpr (IsCarbonSet<typename SetWrapper<SetT>::SetT>) {
|
|
ReportTableMetrics(s_wrapper.s, state);
|
|
}
|
|
}
|
|
|
|
// NOLINTBEGIN(bugprone-macro-parentheses): Parentheses are incorrect here.
|
|
#define MAP_BENCHMARK_ONE_OP_SIZE(NAME, APPLY, KT) \
|
|
BENCHMARK(NAME<Set<KT>>)->Apply(APPLY); \
|
|
BENCHMARK(NAME<absl::flat_hash_set<KT>>)->Apply(APPLY); \
|
|
BENCHMARK(NAME<llvm::DenseSet<KT>>)->Apply(APPLY); \
|
|
BENCHMARK(NAME<llvm::DenseSet<KT, CarbonHashDI<KT>>>)->Apply(APPLY)
|
|
// NOLINTEND(bugprone-macro-parentheses)
|
|
|
|
#define MAP_BENCHMARK_ONE_OP(NAME, APPLY) \
|
|
MAP_BENCHMARK_ONE_OP_SIZE(NAME, APPLY, int); \
|
|
MAP_BENCHMARK_ONE_OP_SIZE(NAME, APPLY, int*); \
|
|
MAP_BENCHMARK_ONE_OP_SIZE(NAME, APPLY, llvm::StringRef)
|
|
|
|
// Benchmark the "latency" of testing for a key in a set. This always tests with
|
|
// a key that is found.
|
|
//
|
|
// However, because the key is always found and because the test ultimately
|
|
// involves conditional control flow that can be predicted, we expect modern
|
|
// CPUs to perfectly predict the control flow here and turn the measurement from
|
|
// one iteration to the next into a throughput measurement rather than a real
|
|
// latency measurement.
|
|
//
|
|
// However, this does represent a particularly common way in which a set data
|
|
// structure is accessed. The numbers should just be carefully interpreted in
|
|
// the context of being more a reflection of reciprocal throughput than actual
|
|
// latency. See the `Lookup` benchmarks for a genuine latency measure with its
|
|
// own caveats.
|
|
//
|
|
// However, this does still show some interesting caching effects when querying
|
|
// large fractions of large tables, and can give a sense of the inescapable
|
|
// magnitude of these effects even when there is a great deal of prediction and
|
|
// speculative execution to hide memory access latency.
|
|
template <typename SetT>
|
|
static void BM_SetContainsHitPtr(benchmark::State& state) {
|
|
using SetWrapperT = SetWrapper<SetT>;
|
|
using KT = SetWrapperT::KeyT;
|
|
SetWrapperT s;
|
|
auto [keys, lookup_keys] =
|
|
GetKeysAndHitKeys<KT>(state.range(0), state.range(1));
|
|
for (auto k : keys) {
|
|
s.BenchInsert(k);
|
|
}
|
|
ssize_t lookup_keys_size = lookup_keys.size();
|
|
|
|
while (state.KeepRunningBatch(lookup_keys_size)) {
|
|
for (ssize_t i = 0; i < lookup_keys_size;) {
|
|
// We block optimizing `i` as that has proven both more effective at
|
|
// blocking the loop from being optimized away and avoiding disruption of
|
|
// the generated code that we're benchmarking.
|
|
benchmark::DoNotOptimize(i);
|
|
|
|
bool result = s.BenchContains(lookup_keys[i]);
|
|
CARBON_DCHECK(result);
|
|
// We use the lookup success to step through keys, establishing a
|
|
// dependency between each lookup. This doesn't fully allow us to measure
|
|
// latency rather than throughput, as noted above.
|
|
i += static_cast<ssize_t>(result);
|
|
}
|
|
}
|
|
}
|
|
MAP_BENCHMARK_ONE_OP(BM_SetContainsHitPtr, HitArgs);
|
|
|
|
// Benchmark the "latency" (but more likely the reciprocal throughput, see
|
|
// comment above) of testing for a key in the set that is *not* present.
|
|
template <typename SetT>
|
|
static void BM_SetContainsMissPtr(benchmark::State& state) {
|
|
using SetWrapperT = SetWrapper<SetT>;
|
|
using KT = SetWrapperT::KeyT;
|
|
SetWrapperT s;
|
|
auto [keys, lookup_keys] = GetKeysAndMissKeys<KT>(state.range(0));
|
|
for (auto k : keys) {
|
|
s.BenchInsert(k);
|
|
}
|
|
ssize_t lookup_keys_size = lookup_keys.size();
|
|
|
|
while (state.KeepRunningBatch(lookup_keys_size)) {
|
|
for (ssize_t i = 0; i < lookup_keys_size;) {
|
|
benchmark::DoNotOptimize(i);
|
|
|
|
bool result = s.BenchContains(lookup_keys[i]);
|
|
CARBON_DCHECK(!result);
|
|
i += static_cast<ssize_t>(!result);
|
|
}
|
|
}
|
|
}
|
|
MAP_BENCHMARK_ONE_OP(BM_SetContainsMissPtr, SizeArgs);
|
|
|
|
// A somewhat contrived latency test for the lookup code path.
|
|
//
|
|
// While lookups into a set are often (but not always) simply used to influence
|
|
// control flow, that style of access produces difficult to evaluate benchmark
|
|
// results (see the comments on the `Contains` benchmarks above).
|
|
//
|
|
// So here we actually access the key in the set and convert that key's value to
|
|
// a boolean on the critical path of each iteration. This lets us have a genuine
|
|
// latency benchmark of looking up a key in the set, at the expense of being
|
|
// somewhat contrived. That said, for usage where the key object is queried or
|
|
// operated on in some way once looked up in the set, this will be fairly
|
|
// representative of the latency cost from the data structure.
|
|
template <typename SetT>
|
|
static void BM_SetLookupHitPtr(benchmark::State& state) {
|
|
using SetWrapperT = SetWrapper<SetT>;
|
|
using KT = SetWrapperT::KeyT;
|
|
SetWrapperT s;
|
|
auto [keys, lookup_keys] =
|
|
GetKeysAndHitKeys<KT>(state.range(0), state.range(1));
|
|
for (auto k : keys) {
|
|
s.BenchInsert(k);
|
|
}
|
|
ssize_t lookup_keys_size = lookup_keys.size();
|
|
|
|
while (state.KeepRunningBatch(lookup_keys_size)) {
|
|
for (ssize_t i = 0; i < lookup_keys_size;) {
|
|
benchmark::DoNotOptimize(i);
|
|
|
|
bool result = s.BenchLookup(lookup_keys[i]);
|
|
CARBON_DCHECK(result);
|
|
i += static_cast<ssize_t>(result);
|
|
}
|
|
}
|
|
}
|
|
MAP_BENCHMARK_ONE_OP(BM_SetLookupHitPtr, HitArgs);
|
|
|
|
// First erase and then insert the key. The code path will always be the same
|
|
// here and so we expect this to largely be a throughput benchmark because of
|
|
// branch prediction and speculative execution.
|
|
//
|
|
// We don't expect erase followed by insertion to be a common user code
|
|
// sequence, but we don't have a good way of benchmarking either erase or insert
|
|
// in isolation -- each would change the size of the table and thus the next
|
|
// iteration's benchmark. And if we try to correct the table size outside of the
|
|
// timed region, we end up trying to exclude too fine grained of a region from
|
|
// timers to get good measurement data.
|
|
//
|
|
// Our solution is to benchmark both erase and insertion back to back. We can
|
|
// then get a good profile of the code sequence of each, and at least measure
|
|
// the sum cost of these reliably. Careful profiling can help attribute that
|
|
// cost between erase and insert in order to understand which of the two
|
|
// operations is contributing most to any performance artifacts observed.
|
|
template <typename SetT>
|
|
static void BM_SetEraseInsertHitPtr(benchmark::State& state) {
|
|
using SetWrapperT = SetWrapper<SetT>;
|
|
using KT = SetWrapperT::KeyT;
|
|
SetWrapperT s;
|
|
auto [keys, lookup_keys] =
|
|
GetKeysAndHitKeys<KT>(state.range(0), state.range(1));
|
|
for (auto k : keys) {
|
|
s.BenchInsert(k);
|
|
}
|
|
ssize_t lookup_keys_size = lookup_keys.size();
|
|
|
|
while (state.KeepRunningBatch(lookup_keys_size)) {
|
|
for (ssize_t i = 0; i < lookup_keys_size;) {
|
|
benchmark::DoNotOptimize(i);
|
|
|
|
s.BenchErase(lookup_keys[i]);
|
|
benchmark::ClobberMemory();
|
|
|
|
bool inserted = s.BenchInsert(lookup_keys[i]);
|
|
CARBON_DCHECK(inserted);
|
|
i += static_cast<ssize_t>(inserted);
|
|
}
|
|
}
|
|
}
|
|
MAP_BENCHMARK_ONE_OP(BM_SetEraseInsertHitPtr, HitArgs);
|
|
|
|
// NOLINTBEGIN(bugprone-macro-parentheses): Parentheses are incorrect here.
|
|
#define MAP_BENCHMARK_OP_SEQ_SIZE(NAME, KT) \
|
|
BENCHMARK(NAME<Set<KT>>)->Apply(SizeArgs); \
|
|
BENCHMARK(NAME<absl::flat_hash_set<KT>>)->Apply(SizeArgs); \
|
|
BENCHMARK(NAME<llvm::DenseSet<KT>>)->Apply(SizeArgs); \
|
|
BENCHMARK(NAME<llvm::DenseSet<KT, CarbonHashDI<KT>>>)->Apply(SizeArgs)
|
|
// NOLINTEND(bugprone-macro-parentheses)
|
|
|
|
#define MAP_BENCHMARK_OP_SEQ(NAME) \
|
|
MAP_BENCHMARK_OP_SEQ_SIZE(NAME, int); \
|
|
MAP_BENCHMARK_OP_SEQ_SIZE(NAME, int*); \
|
|
MAP_BENCHMARK_OP_SEQ_SIZE(NAME, llvm::StringRef)
|
|
|
|
// This is an interesting, somewhat specialized benchmark that measures the cost
|
|
// of inserting a sequence of keys into a set up to some size and then inserting
|
|
// a colliding key and throwing away the set.
|
|
//
|
|
// This is an especially important usage pattern for sets as a large number of
|
|
// algorithms essentially look like this, such as collision detection, cycle
|
|
// detection, de-duplication, etc.
|
|
//
|
|
// It also covers both the insert-into-an-empty-slot code path that isn't
|
|
// covered elsewhere, and the code path for growing a table to a larger size.
|
|
//
|
|
// This is the second most important aspect of expected set usage after testing
|
|
// for presence. It also nicely lends itself to a single benchmark that covers
|
|
// the total cost of this usage pattern.
|
|
//
|
|
// Because this benchmark operates on whole sets, we also compute the number of
|
|
// probed keys for Carbon's set as that is both a general reflection of the
|
|
// efficacy of the underlying hash function, and a direct factor that drives the
|
|
// cost of these operations.
|
|
template <typename SetT>
|
|
static void BM_SetInsertSeq(benchmark::State& state) {
|
|
using SetWrapperT = SetWrapper<SetT>;
|
|
using KT = SetWrapperT::KeyT;
|
|
constexpr ssize_t LookupKeysSize = 1 << 8;
|
|
auto [keys, lookup_keys] =
|
|
GetKeysAndHitKeys<KT>(state.range(0), LookupKeysSize);
|
|
|
|
// Now build a large shuffled set of keys (with duplicates) we'll use at the
|
|
// end.
|
|
ssize_t i = 0;
|
|
for (auto _ : state) {
|
|
benchmark::DoNotOptimize(i);
|
|
|
|
SetWrapperT s;
|
|
for (auto k : keys) {
|
|
bool inserted = s.BenchInsert(k);
|
|
CARBON_DCHECK(inserted, "Must be a successful insert!");
|
|
}
|
|
|
|
// Now insert a final random repeated key.
|
|
bool inserted = s.BenchInsert(lookup_keys[i]);
|
|
CARBON_DCHECK(!inserted, "Must already be in the map!");
|
|
|
|
// Rotate through the shuffled keys.
|
|
i = (i + static_cast<ssize_t>(!inserted)) & (LookupKeysSize - 1);
|
|
}
|
|
|
|
// It can be easier in some cases to think of this as a key-throughput rate of
|
|
// insertion rather than the latency of inserting N keys, so construct the
|
|
// rate counter as well.
|
|
state.counters["KeyRate"] = benchmark::Counter(
|
|
keys.size(), benchmark::Counter::kIsIterationInvariantRate);
|
|
|
|
// Report some extra statistics about the Carbon type.
|
|
if constexpr (IsCarbonSet<SetT>) {
|
|
// Re-build a set outside of the timing loop to look at the statistics
|
|
// rather than the timing.
|
|
SetT s;
|
|
for (auto k : keys) {
|
|
bool inserted = s.Insert(k).is_inserted();
|
|
CARBON_DCHECK(inserted, "Must be a successful insert!");
|
|
}
|
|
|
|
ReportTableMetrics(s, state);
|
|
|
|
// Uncomment this call to print out statistics about the index-collisions
|
|
// among these keys for debugging:
|
|
//
|
|
// RawHashtable::DumpHashStatistics(raw_keys);
|
|
}
|
|
}
|
|
MAP_BENCHMARK_OP_SEQ(BM_SetInsertSeq);
|
|
|
|
// Benchmark visiting every key in a set.
|
|
//
|
|
// Unlike the lookup benchmarks, this walks the table's storage from end to end
|
|
// rather than probing it, so it is largely a measure of how densely keys are
|
|
// packed and how cheaply empty slots can be skipped. There is no dependency
|
|
// between the keys visited, and so this is a throughput measurement.
|
|
//
|
|
// Each batch is a single complete traversal of the set, with the batch size set
|
|
// to the number of keys so that the reported time is the per-key cost.
|
|
template <typename SetT>
|
|
static void BM_SetIterate(benchmark::State& state) {
|
|
using SetWrapperT = SetWrapper<SetT>;
|
|
using KT = typename SetWrapperT::KeyT;
|
|
SetWrapperT s;
|
|
auto [keys, _] = GetKeysAndMissKeys<KT>(state.range(0));
|
|
for (auto k : keys) {
|
|
bool inserted = s.BenchInsert(k);
|
|
CARBON_DCHECK(inserted, "Must be a successful insert!");
|
|
}
|
|
|
|
while (state.KeepRunningBatch(keys.size())) {
|
|
ssize_t sum = 0;
|
|
s.BenchIterate([&sum](const KT& k) {
|
|
// Consume the key so that neither the traversal nor the loads out of the
|
|
// entries can be optimized away.
|
|
sum += ValueToBool(k);
|
|
});
|
|
benchmark::DoNotOptimize(sum);
|
|
}
|
|
|
|
// The time is already per-key, so an iteration-invariant rate of one gives
|
|
// the throughput of keys visited.
|
|
state.counters["KeyRate"] =
|
|
benchmark::Counter(1, benchmark::Counter::kIsIterationInvariantRate);
|
|
|
|
ReportMetrics(s, state);
|
|
}
|
|
MAP_BENCHMARK_ONE_OP(BM_SetIterate, SizeArgs);
|
|
|
|
} // namespace
|
|
} // namespace Carbon
|