Files
Chandler Carruth 632a81fcb6 Keep benchmark values passed to DoNotOptimize in registers (#7870)
Clang implements the `"+r,m"` constraint in `benchmark::DoNotOptimize`
by always choosing memory, so each call stores the value to the stack
and loads it back. Most of our benchmarks call it on a loop counter or
another value carried to the next iteration, which puts that store and
reload on the loop's critical path.

On an Apple M1, the cost of that round trip depends on which register
the compiler uses to address the stack slot, and unrelated code changes
move it. Changing only that register from `x29` to `sp`, with the same
address, made `BM_SetLookupHitPtr<Set<int>>` 41.5% to 49.1% slower.

Add `Carbon::Testing::DoNotOptimize` in
`//testing/base:benchmark_helpers`, and switch every benchmark to it. It
only accepts types it can keep in registers: integers, enums, and
pointers go into a `"+r"` constraint with a `"memory"` clobber, and
containers pass their `data()` pointer to that form, so the compiler
must assume the call reads and writes their contents. Any other type is
a compile error. The container form doesn't block optimizing based on a
container's size; passing the container's address does.

Benchmarks that pass a loop counter or carried value no longer measure a
store and reload per iteration, so their numbers aren't comparable with
earlier runs. For example, the hashing latency benchmarks no longer
include one in each hash's latency.

Assisted-by: Claude Code
2026-10-02 00:15:21 +00:00

122 lines
2.9 KiB
Python

# Part of the Carbon Language project, under the Apache License v2.0 with LLVM
# Exceptions. See /LICENSE for license information.
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#
# Trivial, single-file testing libraries. More complex libraries should get
# their own directory.
load("//bazel/cc_rules:defs.bzl", "cc_library", "cc_test")
package(default_visibility = ["//visibility:public"])
# This does extra initialization on top of googletest's gtest_main in order to
# provide stack traces on unexpected exits, because we normally rely on LLVM
# code for that.
#
# This replaces "@googletest//:gtest_main";
# "@googletest//:gtest" should still be used directly.
cc_library(
name = "gtest_main",
testonly = 1,
srcs = ["gtest_main.cpp"],
deps = [
":global_exe_path",
"//common:init_llvm",
"@googletest//:gtest",
"@llvm-project//llvm:Support",
],
)
# This does extra initialization on top of Google benchmark's main in order to
# provide stack traces and setup LLVM.
#
# This replaces `@google_benchmark//:benchmark_main`;
# `@google_benchmark//:benchmark` should still be used directly.
cc_library(
name = "benchmark_main",
testonly = 1,
srcs = ["benchmark_main.cpp"],
deps = [
":global_exe_path",
"//common:init_llvm",
"@abseil-cpp//absl/flags:parse",
"@google_benchmark//:benchmark",
"@llvm-project//llvm:Support",
],
)
cc_library(
name = "benchmark_helpers",
testonly = 1,
hdrs = ["benchmark_helpers.h"],
)
cc_library(
name = "capture_std_streams",
testonly = 1,
srcs = ["capture_std_streams.cpp"],
hdrs = ["capture_std_streams.h"],
deps = [
"//common:ostream",
"@googletest//:gtest",
],
)
cc_library(
name = "file_helpers",
testonly = 1,
srcs = ["file_helpers.cpp"],
hdrs = ["file_helpers.h"],
deps = [
"//common:error",
"@googletest//:gtest",
],
)
cc_library(
name = "global_exe_path",
testonly = 1,
srcs = ["global_exe_path.cpp"],
hdrs = ["global_exe_path.h"],
deps = [
"//common:check",
"//common:exe_path",
"@llvm-project//llvm:Support",
],
)
cc_test(
name = "global_exe_path_test",
size = "small",
srcs = ["global_exe_path_test.cpp"],
deps = [
":global_exe_path",
":gtest_main",
"@googletest//:gtest",
"@llvm-project//llvm:Support",
],
)
cc_library(
name = "unified_diff",
testonly = 1,
hdrs = ["unified_diff.h"],
deps = [
"//common:check",
"@googletest//:gtest",
"@llvm-project//llvm:Support",
],
)
cc_test(
name = "unified_diff_test",
size = "small",
srcs = ["unified_diff_test.cpp"],
deps = [
":gtest_main",
":unified_diff",
"@googletest//:gtest",
"@llvm-project//llvm:Support",
],
)