Add a verifier that SemIR is in SSA form. (#7771)

Check that every use of a non-constant instruction is dominated by a
definition of that instruction. Remove the fake (instruction creation
order based) dominance checks in convert; these start spuriously failing
during template instantiation of initializers.

Assisted-by: Claude Opus 5 via Antigravity
This commit is contained in:
Richard Smith
2026-09-29 20:49:37 +00:00
committed by GitHub
parent 5dae7b40b9
commit c2d9cde674
16 changed files with 2006 additions and 104 deletions
+22
View File
@@ -93,6 +93,28 @@ cc_library(
hdrs = ["for_each_macro.h"],
)
cc_library(
name = "grouped_value_store",
hdrs = ["grouped_value_store.h"],
deps = [
":id_tag",
"//common:check",
"@llvm-project//llvm:Support",
],
)
cc_test(
name = "grouped_value_store_test",
size = "small",
srcs = ["grouped_value_store_test.cpp"],
deps = [
":grouped_value_store",
":index_base",
"//testing/base:gtest_main",
"@googletest//:gtest",
],
)
cc_library(
name = "id_tag",
hdrs = ["id_tag.h"],
+150
View File
@@ -0,0 +1,150 @@
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#ifndef CARBON_TOOLCHAIN_BASE_GROUPED_VALUE_STORE_H_
#define CARBON_TOOLCHAIN_BASE_GROUPED_VALUE_STORE_H_
#include <concepts>
#include <cstddef>
#include <cstdint>
#include "common/check.h"
#include "llvm/ADT/ArrayRef.h"
#include "llvm/ADT/Sequence.h"
#include "llvm/ADT/SmallVector.h"
#include "toolchain/base/id_tag.h"
namespace Carbon {
// A fixed collection of values, grouped by an ID: for example, the specifics of
// each generic, or the instructions checked from each token.
//
// The values are held in a single array, ordered by the ID of the group they
// belong to, alongside the index at which each group starts. Because IDs are
// dense, that layout can be built by a counting sort in two passes over the
// values, and a group can then be found in constant time. A map from ID to a
// list of values would instead hash on every lookup and allocate per group.
//
// The trade-off is that the contents are fixed once built: a group is a
// contiguous range of the value array, so nothing can be added to it later.
template <typename IdT, typename ValueT, typename TagIdT = Untagged>
class GroupedValueStore {
public:
using IdType = IdT;
using IdTagType = IdTag<IdT, TagIdT>;
// Builds the groups for the IDs vended by `id_source`, which is the value
// store that the group IDs index.
//
// `enumerate` is called with a function to call with each (ID, value) pair to
// store; pairs whose ID has no value are ignored. Note that `enumerate` is
// called more than once, and must produce the same pairs each time.
template <typename ValueStoreT, typename EnumerateFn>
requires(std::same_as<IdT, typename ValueStoreT::IdType> &&
!IdTagIsUntagged<IdTagType>)
explicit GroupedValueStore(const ValueStoreT& id_source,
EnumerateFn enumerate)
: GroupedValueStore(id_source.size(), id_source.GetIdTag(), enumerate) {}
// Builds the groups for the `num_ids` IDs with indexes `[0, num_ids)`.
template <typename EnumerateFn>
requires(IdTagIsUntagged<IdTagType>)
explicit GroupedValueStore(size_t num_ids, EnumerateFn enumerate)
: GroupedValueStore(num_ids, IdTagType(), enumerate) {}
// Returns the values in the group for `id`, in the order they were
// enumerated. This is empty if no value was added for `id`, including when
// `id` has no value.
auto Get(IdT id) const -> llvm::ArrayRef<ValueT> {
if (!id.has_value()) {
return {};
}
int32_t index = tag_.Remove(id);
CARBON_CHECK(static_cast<size_t>(index) + 1 < starts_.size(),
"{0} is not an ID of a group in this store", id);
return llvm::ArrayRef(values_).slice(starts_[index],
starts_[index + 1] - starts_[index]);
}
auto size() const -> size_t { return values_.size(); }
private:
template <typename EnumerateFn>
explicit GroupedValueStore(size_t num_ids, IdTagType tag,
EnumerateFn enumerate);
// The values of every group, ordered by group: the group for the ID with
// index `i` is `values_[starts_[i] .. starts_[i + 1])`, so that `starts_` has
// one more element than there are IDs.
llvm::SmallVector<ValueT, 0> values_;
llvm::SmallVector<int32_t, 0> starts_;
IdTagType tag_;
};
template <typename IdT, typename ValueT, typename TagIdT>
template <typename EnumerateFn>
GroupedValueStore<IdT, ValueT, TagIdT>::GroupedValueStore(size_t num_ids,
IdTagType tag,
EnumerateFn enumerate)
: tag_(tag) {
// The index of the group that `id` belongs to. This is checked rather than
// assumed because an out-of-range index would write outside `starts_`.
auto group_index = [&](IdT id) {
int32_t index = tag_.Remove(id);
CARBON_CHECK(static_cast<size_t>(index) < num_ids,
"{0} is not an ID of a group in this store", id);
return index;
};
// Populate `starts_` in three in-place passes. Note that we need N+1 elements
// to hold the boundaries of N contiguous groups, plus an additional temporary
// element for reasons discussed below.
//
// First, we count the values per ID. The array contents are shifted by 2:
// `starts_[i + 2]` will hold the number of values for the ID with `.index ==
// i`.
starts_.assign(num_ids + 2, 0);
enumerate([&](IdT id, const ValueT& /*value*/) {
if (id.has_value()) {
++starts_[group_index(id) + 2];
}
});
// Perform a prefix sum, so that `starts_[i + 2]` holds the number of values
// for IDs with `.index <= i`, that is, the end of the group for ID `i`, and
// hence `starts_[i + 1]` is the start of the group for ID `i`.
for (auto i : llvm::seq<size_t>(1, starts_.size())) {
starts_[i] += starts_[i - 1];
}
// Pop the final "start" index, which is now the total number of values in all
// the groups.
int32_t num_values = starts_.pop_back_val();
// Every element is written below, but the array still has to be filled with
// something first, and an ID type has no default constructor.
if constexpr (requires { ValueT::None; }) {
values_.resize(num_values, ValueT::None);
} else {
values_.resize(num_values);
}
// Populate `values_`, using `starts_[i + 1]` as the index to write the next
// value for ID `i`, which is incremented on each write. Thus, at the end of
// the loop, `starts_[i + 1]` is the past-the-end index for ID `i`, that is,
// the start index for ID `i + 1`, which is the final state of `starts_`.
enumerate([&](IdT id, const ValueT& value) {
if (id.has_value()) {
values_[starts_[group_index(id) + 1]++] = value;
}
});
CARBON_CHECK(static_cast<size_t>(starts_.back()) == values_.size(),
"Enumeration produced a different number of values each time");
}
} // namespace Carbon
#endif // CARBON_TOOLCHAIN_BASE_GROUPED_VALUE_STORE_H_
@@ -0,0 +1,75 @@
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include "toolchain/base/grouped_value_store.h"
#include <gmock/gmock.h>
#include <gtest/gtest.h>
#include <utility>
#include "toolchain/base/index_base.h"
namespace Carbon::Testing {
namespace {
using ::testing::ElementsAre;
using ::testing::IsEmpty;
struct TestId : public IndexBase<TestId> {
// Only used when a `TestId` is printed, for example by a failing matcher.
[[maybe_unused]] static constexpr llvm::StringLiteral Label = "test";
using IndexBase::IndexBase;
};
TEST(GroupedValueStore, Empty) {
GroupedValueStore<TestId, int> store(0, [](auto /*add*/) {});
EXPECT_EQ(store.size(), 0U);
EXPECT_THAT(store.Get(TestId::None), IsEmpty());
}
TEST(GroupedValueStore, GroupsValuesById) {
std::pair<int, char> values[] = {{2, 'a'}, {0, 'b'}, {2, 'c'}};
GroupedValueStore<TestId, char> store(4, [&](auto add) {
for (auto [index, value] : values) {
add(TestId(index), value);
}
});
EXPECT_EQ(store.size(), 3U);
EXPECT_THAT(store.Get(TestId(0)), ElementsAre('b'));
// A group with no values is empty, not missing.
EXPECT_THAT(store.Get(TestId(1)), IsEmpty());
// Values within a group are in the order they were enumerated.
EXPECT_THAT(store.Get(TestId(2)), ElementsAre('a', 'c'));
EXPECT_THAT(store.Get(TestId(3)), IsEmpty());
}
TEST(GroupedValueStore, IgnoresIdWithNoValue) {
GroupedValueStore<TestId, char> store(2, [](auto add) {
add(TestId::None, 'a');
add(TestId(1), 'b');
});
EXPECT_EQ(store.size(), 1U);
EXPECT_THAT(store.Get(TestId::None), IsEmpty());
EXPECT_THAT(store.Get(TestId(1)), ElementsAre('b'));
}
TEST(GroupedValueStore, StoresIdValues) {
// An ID type has no default constructor, so this exercises filling the value
// array with `None` before overwriting it.
GroupedValueStore<TestId, TestId> store(2, [](auto add) {
add(TestId(0), TestId(7));
add(TestId(0), TestId(8));
});
EXPECT_THAT(store.Get(TestId(0)), ElementsAre(TestId(7), TestId(8)));
EXPECT_THAT(store.Get(TestId(1)), IsEmpty());
}
} // namespace
} // namespace Carbon::Testing
-8
View File
@@ -2543,14 +2543,6 @@ auto InitializeExisting(Context& context, SemIR::LocId loc_id,
storage_id = SemIR::InstId::None;
}
// TODO: This is only an approximation of a dominance check. Add a general
// end-of-phase dominance check and remove the check here and the one in
// `MergeReplacing`.
CARBON_CHECK(!storage_id.has_value() ||
value_id == SemIR::ErrorInst::InstId ||
context.insts().GetRawIndex(storage_id) <=
context.insts().GetRawIndex(value_id),
"Storage might not dominate initializer");
PendingBlock target_block(&context);
return Convert(context, loc_id, value_id,
{.kind = ConversionTarget::Initializing,
+3 -7
View File
@@ -152,13 +152,9 @@ struct InitializeResult {
// be inserted before any use of the storage by the initializer, and will be
// inserted even if the initializer does not actually use the storage. It must
// be valid to reference `storage_id` after splicing in `storage_access_block`,
// so `storage_id` must either dominate the initializer (but see the TODO below)
// or be one of the instructions in `storage_access_block`. If `storage_id` is
// known to always dominate the initializer, `InitializeExisting` should be used
// instead.
//
// TODO: We don't have an implementation of a proper dominance check, so we
// fake one up by comparing the order in which the insts were created.
// so `storage_id` must either dominate the initializer or be one of the
// instructions in `storage_access_block`. If `storage_id` is known to always
// dominate the initializer, `InitializeExisting` should be used instead.
//
// This function does not guarantee to perform an in-place initialization, so
// the caller is responsible for passing the returned `inst_id` to an inst that
+2 -11
View File
@@ -84,20 +84,11 @@ class PendingBlock {
// Replace the instruction at target_id with the instructions in this block.
// The new value for target_id should be value_id. Returns the InstId that
// should be used to refer to the result from now on. value_id must dominate
// target_id (but see below), or refer to an instruction within this block, in
// order to preserve the property that SemIR is topologically sorted.
//
// TODO: We don't have an implementation of a proper dominance check, so we
// fake one up by comparing the order in which the insts were created.
// Add a general end-of-phase dominance check and remove the one here and in
// `InitializeExisting`.
// target_id, or refer to an instruction within this block, in order to
// preserve the property that SemIR is topologically sorted.
auto MergeReplacing(SemIR::InstId target_id, SemIR::InstId value_id)
-> SemIR::InstId {
CARBON_CHECK(target_id != value_id);
CARBON_CHECK(context_->insts().GetRawIndex(value_id) <=
context_->insts().GetRawIndex(target_id) ||
llvm::is_contained(insts_, value_id),
"Splice might break dominance condition");
SemIR::LocIdAndInst value = context_->insts().GetWithLocId(value_id);
auto result_id = value_id;
+1 -1
View File
@@ -32,7 +32,7 @@ cc_library(
srcs = ["sem_ir_index.cpp"],
hdrs = ["sem_ir_index.h"],
deps = [
"//common:check",
"//toolchain/base:grouped_value_store",
"//toolchain/lex:token_index",
"//toolchain/lex:token_kind",
"//toolchain/lex:tokenized_buffer",
+6 -59
View File
@@ -4,7 +4,6 @@
#include "toolchain/language_server/sem_ir_index.h"
#include "common/check.h"
#include "toolchain/lex/token_kind.h"
#include "toolchain/parse/node_kind.h"
#include "toolchain/parse/tree.h"
@@ -52,63 +51,11 @@ static auto GetTokenForInst(const SemIR::File& sem_ir,
}
SemIRIndex::SemIRIndex(const SemIR::File& sem_ir,
const Parse::TreeAndSubtrees& tree_and_subtrees) {
const auto& tokens = tree_and_subtrees.tree().tokens();
// Populate `token_starts_` in three in-place passes. Note that we need N+1
// elements to hold the boundaries of N contiguous intervals, plus an
// additional temporary element for reasons discussed below.
//
// First, we count the instructions per token. The array contents are shifted
// by 2: `token_starts[i+2]` will hold the number of insts for the token with
// `.index == i`.
token_starts_.assign(tokens.size() + 2, 0);
for (auto [inst_id, inst] : sem_ir.insts().enumerate()) {
auto token = GetTokenForInst(sem_ir, tree_and_subtrees, inst_id);
if (!token.has_value()) {
continue;
}
++token_starts_[token.index + 2];
}
// Perform a prefix sum, so that `token_starts_[i+2]` holds the number of
// insts for tokens with `.index <= i`, i.e. the end of the interval for token
// `i`, and hence `token_starts_[i+1]` is the start of the interval for token
// `i`.
for (size_t i = 1; i < token_starts_.size(); ++i) {
token_starts_[i] += token_starts_[i - 1];
}
// Pop the final "start" index, which is now the total number of instructions
// that have associated locations.
auto total_insts = token_starts_.pop_back_val();
// Populate `insts_`, using `token_starts_[i+1]` as the index to write the
// next inst for token `i`, which is incremented on each write. Thus, at the
// end of the loop, `token_starts_[i+1]` is the past-the-end index for token
// `i`, i.e. the start index for token `i+1`, which is the final state of
// `token_starts_`.
insts_.resize(total_insts, SemIR::InstId::None);
for (auto [inst_id, inst] : sem_ir.insts().enumerate()) {
auto token = GetTokenForInst(sem_ir, tree_and_subtrees, inst_id);
if (!token.has_value()) {
continue;
}
insts_[token_starts_[token.index + 1]++] = inst_id;
}
CARBON_CHECK(static_cast<size_t>(token_starts_.back()) == insts_.size());
}
auto SemIRIndex::InstsForToken(Lex::TokenIndex token) const
-> llvm::ArrayRef<SemIR::InstId> {
if (!token.has_value()) {
return {};
}
CARBON_CHECK(static_cast<size_t>(token.index) + 1 < token_starts_.size(),
"Token {0} is not from the indexed file", token.index);
int32_t start = token_starts_[token.index];
int32_t end = token_starts_[token.index + 1];
return llvm::ArrayRef(insts_).slice(start, end - start);
}
const Parse::TreeAndSubtrees& tree_and_subtrees)
: insts_(tree_and_subtrees.tree().tokens().size(), [&](auto add) {
for (auto [inst_id, inst] : sem_ir.insts().enumerate()) {
add(GetTokenForInst(sem_ir, tree_and_subtrees, inst_id), inst_id);
}
}) {}
} // namespace Carbon::LanguageServer
+10 -12
View File
@@ -6,7 +6,7 @@
#define CARBON_TOOLCHAIN_LANGUAGE_SERVER_SEM_IR_INDEX_H_
#include "llvm/ADT/ArrayRef.h"
#include "llvm/ADT/SmallVector.h"
#include "toolchain/base/grouped_value_store.h"
#include "toolchain/lex/token_index.h"
#include "toolchain/lex/tokenized_buffer.h"
#include "toolchain/parse/node_ids.h"
@@ -49,19 +49,17 @@ class SemIRIndex {
// punctuation and keywords usually contribute to an enclosing instruction
// rather than producing one of their own.
auto InstsForToken(Lex::TokenIndex token) const
-> llvm::ArrayRef<SemIR::InstId>;
-> llvm::ArrayRef<SemIR::InstId> {
return insts_.Get(token);
}
private:
// Instructions grouped by token, in the compressed-sparse-row layout: the
// group for token `i` is `insts_[token_starts_[i] .. token_starts_[i + 1])`.
// `token_starts_` therefore has one more entry than there are tokens.
//
// Token indices are dense, so this is built by counting sort in a single pass
// over the instructions, and looked up in constant time. A hash map would
// need to handle the many-instructions-per-token case explicitly; here it
// falls out of the layout.
llvm::SmallVector<SemIR::InstId, 0> insts_;
llvm::SmallVector<int32_t, 0> token_starts_;
// Instructions grouped by the token they were checked from. Token indices are
// dense, so this is built by counting sort in a single pass over the
// instructions, and looked up in constant time. A hash map would need to
// handle the many-instructions-per-token case explicitly; here it falls out
// of the layout.
GroupedValueStore<Lex::TokenIndex, SemIR::InstId> insts_;
};
} // namespace Carbon::LanguageServer
+48 -1
View File
@@ -2,7 +2,7 @@
# Exceptions. See /LICENSE for license information.
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
load("//bazel/cc_rules:defs.bzl", "cc_library", "cc_test")
load("//bazel/cc_rules:defs.bzl", "cc_binary", "cc_library", "cc_test")
package(default_visibility = ["//visibility:public"])
@@ -100,6 +100,7 @@ cc_library(
"cpp_initializer_list.cpp",
"cpp_overload_set.cpp",
"declared_facet_type.cpp",
"dominance.cpp",
"entity_name.cpp",
"field.cpp",
"file.cpp",
@@ -135,6 +136,7 @@ cc_library(
"cpp_initializer_list.h",
"cpp_overload_set.h",
"declared_facet_type.h",
"dominance.h",
"entity_name.h",
"entity_with_params_base.h",
"field.h",
@@ -180,6 +182,7 @@ cc_library(
"//common:struct_reflection",
"//toolchain/base:block_value_store",
"//toolchain/base:canonical_value_store",
"//toolchain/base:grouped_value_store",
"//toolchain/base:id_tag",
"//toolchain/base:index_base",
"//toolchain/base:int",
@@ -379,6 +382,50 @@ cc_test(
],
)
cc_library(
name = "dominance_test_helpers",
testonly = 1,
hdrs = ["dominance_test_helpers.h"],
deps = [
":file",
":typed_insts",
"//toolchain/base:shared_value_stores",
"//toolchain/parse:node_kind",
"@llvm-project//llvm:Support",
],
)
cc_test(
name = "dominance_test",
size = "small",
srcs = ["dominance_test.cpp"],
deps = [
":dominance_test_helpers",
":file",
":typed_insts",
"//testing/base:gtest_main",
"//toolchain/parse:node_kind",
"@googletest//:gtest",
"@llvm-project//llvm:Support",
],
)
cc_binary(
name = "dominance_benchmark",
testonly = 1,
srcs = ["dominance_benchmark.cpp"],
deps = [
":dominance_test_helpers",
":file",
":typed_insts",
"//common:check",
"//common:error",
"//testing/base:benchmark_main",
"@google_benchmark//:benchmark",
"@llvm-project//llvm:Support",
],
)
cc_test(
name = "yaml_test",
size = "small",
+720
View File
@@ -0,0 +1,720 @@
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include "toolchain/sem_ir/dominance.h"
#include <algorithm>
#include <utility>
#include "common/check.h"
#include "common/error.h"
#include "common/map.h"
#include "llvm/ADT/ArrayRef.h"
#include "llvm/ADT/BitVector.h"
#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/Sequence.h"
#include "llvm/ADT/SmallBitVector.h"
#include "llvm/ADT/SmallVector.h"
#include "toolchain/base/grouped_value_store.h"
#include "toolchain/base/id_tag.h"
#include "toolchain/base/index_base.h"
#include "toolchain/base/kind_switch.h"
#include "toolchain/sem_ir/file.h"
#include "toolchain/sem_ir/function.h"
#include "toolchain/sem_ir/generic.h"
#include "toolchain/sem_ir/id_kind.h"
#include "toolchain/sem_ir/ids.h"
#include "toolchain/sem_ir/inst.h"
#include "toolchain/sem_ir/typed_insts.h"
namespace Carbon::SemIR {
namespace {
// The position of a block within a function's list of body blocks. This is used
// to index the control flow graph and dominator tree built for the function.
struct BlockIndex : public IndexBase<BlockIndex> {
// Only used when a `BlockIndex` is printed, which is useful when debugging.
[[maybe_unused]] static constexpr llvm::StringLiteral Label = "block";
using IndexBase::IndexBase;
};
// Execution of a function body starts in its first block.
constexpr BlockIndex EntryBlockIndex(0);
// A step in the walk over a function's dominator tree.
struct WalkStep {
// Returns a step that verifies the instructions in `block_index` and queues
// up the blocks it dominates.
static auto EnterBlock(BlockIndex block_index) -> WalkStep {
return {.block_index = block_index, .scope_start = 0};
}
// Returns a step that leaves the scope of a block that was entered when the
// list of evaluated instructions had size `scope_start`.
static auto LeaveBlock(int scope_start) -> WalkStep {
return {.block_index = BlockIndex::None, .scope_start = scope_start};
}
// The block to enter, or `None` if this step leaves a block instead.
BlockIndex block_index;
// The number of instructions that had been evaluated when the block being
// left was entered. Only used when leaving a block.
int scope_start;
};
// Returns the block that `inst` transfers control to, or `InstBlockId::None` if
// `inst` is not a branch.
auto GetBranchTargetId(Inst inst) -> InstBlockId {
if (auto branch = inst.TryAs<AnyBranch>()) {
return branch->target_id;
}
return InstBlockId::None;
}
// A set of the instructions in a file, holding one bit per instruction.
//
// Creating one of these costs a bit per instruction in the file, so they are
// created once for the file rather than once per function.
class InstSet {
public:
explicit InstSet(const File& file)
: tag_(file.insts().GetIdTag()), bits_(file.insts().size()) {}
// Adds `inst_id`, returning whether it wasn't already present.
auto Insert(InstId inst_id) -> bool {
int32_t index = tag_.Remove(inst_id);
if (bits_.test(index)) {
return false;
}
bits_.set(index);
return true;
}
auto Erase(InstId inst_id) -> void { bits_.reset(tag_.Remove(inst_id)); }
auto Contains(InstId inst_id) const -> bool {
return bits_.test(tag_.Remove(inst_id));
}
private:
// Instruction IDs are tagged, so the tag is removed to recover the index of
// the instruction, which is the index of its bit.
InstStore::IdTagType tag_;
llvm::BitVector bits_;
};
// Collects the instructions that are referenced from function bodies despite
// being evaluated at file scope, or not evaluated at all, into `decl_insts`.
//
// TODO: These are pre-existing dominance violations, not genuine exemptions:
//
// - File-scope instructions are evaluated in `__global_init`, if at all, so
// they don't dominate uses in another function.
// - A `let` in a class body produces a `wrapper_binding` that isn't evaluated
// in any function, but a qualified name reference to it can appear in one.
// For example, from the `public_global_access` case in
// `check/testdata/class/access/access_modifiers.carbon`:
//
// class A { let x: i32 = 5; }
// let x: i32 = A.x;
//
// Here the `name_ref` for `A.x` is in `__global_init`, and the
// `wrapper_binding` for `x` is only in `A`'s body block.
//
// Lowering only works today because such uses either happen to be constant or
// are never lowered. Remove this allowlist and diagnose the uses instead once
// global initialization semantics and the member reference model are settled.
auto CollectDeclInsts(const File& file, InstSet& decl_insts) -> void {
auto add_block = [&](InstBlockId block_id) {
if (block_id.has_value()) {
for (InstId inst_id : file.inst_blocks().Get(block_id)) {
if (inst_id.has_value()) {
decl_insts.Insert(inst_id);
}
}
}
};
add_block(file.top_inst_block_id());
for (const auto& class_info : file.classes().values()) {
add_block(class_info.body_block_id);
}
}
// The specifics of each generic in a file.
using SpecificsByGeneric =
GroupedValueStore<GenericId, SpecificId, Tag<CheckIRId>>;
// Collects the specifics in `file`, grouped by the generic they're a specific
// of. A specific that has never been resolved, or whose resolution failed,
// doesn't have instructions to check, so is omitted.
//
// This grouping is built once for the file so that each generic function can
// find its own specifics without scanning all of them.
//
// TODO: Consider moving this to `File` or `SpecificStore` if other passes need
// to look up all specifics of a generic.
auto CollectSpecifics(const File& file) -> SpecificsByGeneric {
return SpecificsByGeneric(file.generics(), [&](auto add) {
for (const auto& [specific_id, specific] : file.specifics().enumerate()) {
if (specific.IsUnresolved() || specific.HasError()) {
continue;
}
add(specific.generic_id, specific_id);
}
});
}
// The dominator tree of a function body: the blocks that each block
// immediately dominates, indexed by `BlockIndex`.
// TODO: Explore replacing this with a flattened vector plus a vector of
// starting indexes to reduce the number of heap allocations required and the
// vector overhead.
using DominatorTree = llvm::SmallVector<llvm::SmallVector<BlockIndex, 2>>;
// Builds the control flow graph of a function body and its dominator tree.
//
// These depend only on the branches between the function's body blocks, which
// are the same in every specific of a generic function: a block spliced into
// the body never contains control flow. So the tree is built once per function
// and shared by the verification of each of its specifics.
class DominatorTreeBuilder {
public:
explicit DominatorTreeBuilder(const File& file, const Function& function)
: file_(file),
function_(function),
body_blocks_(function.body_block_ids) {}
// Returns the dominator tree of the function body, diagnosing a branch that
// leaves the body and blocks that are unreachable from the entry block.
auto Build() -> ErrorOr<DominatorTree>;
private:
// Builds `successors_` and `predecessors_` from the branches in each block.
auto BuildControlFlowGraph() -> ErrorOr<Success>;
// Appends the blocks reachable from the entry block to `post_order`, in
// post-order, and marks each of them `visited`.
auto BuildPostOrder(llvm::SmallBitVector& visited,
llvm::SmallVectorImpl<BlockIndex>& post_order) -> void;
auto num_blocks() const -> int { return body_blocks_.size(); }
const File& file_;
const Function& function_;
llvm::ArrayRef<InstBlockId> body_blocks_;
// The index of each block in `body_blocks_`.
Map<InstBlockId, BlockIndex> block_indexes_;
// The control flow graph, indexed by `BlockIndex`.
llvm::SmallVector<llvm::SmallVector<BlockIndex, 2>> successors_;
llvm::SmallVector<llvm::SmallVector<BlockIndex, 2>> predecessors_;
};
// Verifies that every operand of every instruction in one function body is
// dominated by an evaluation of that operand.
//
// This walks the dominator tree of the function's control flow graph, tracking
// the set of instructions whose evaluations dominate the point currently being
// checked. An instruction joins that set when its evaluation is reached, and
// leaves it again when the walk leaves the blocks that the evaluation
// dominates.
//
// TODO: Improve LLVM's GenericDomTree implementation so that it's compatible
// with our graph representation, then rewrite this to use that rather than
// implementing our own dominator tree construction. Currently, GenericDomTree
// requires a pointer-based data structure, and building such a data structure
// introduces a substantial performance overhead compared to running dominator
// tree construction directly on our SemIR representation.
class DominanceVerifier {
public:
explicit DominanceVerifier(const File& file, const InstSet& decl_insts,
const Function& function,
const DominatorTree& dom_tree, InstSet& evaluated,
SpecificId specific_id)
: file_(file),
decl_insts_(decl_insts),
function_(function),
dom_tree_(dom_tree),
specific_id_(specific_id),
body_blocks_(function.body_block_ids),
evaluated_(evaluated) {}
auto Verify() -> ErrorOr<Success>;
private:
// Verifies every block, walking the dominator tree from the entry block.
auto VerifyBlocks() -> ErrorOr<Success>;
// Verifies the operands of `inst_id`, then records `inst_id`, along with any
// instructions it splices into the enclosing block, as evaluated.
auto VerifyAndRecordInst(InstId root_inst_id, BlockIndex block_index)
-> ErrorOr<Success>;
// Verifies a single operand of `user_id`.
auto VerifyArg(InstId user_id, IdAndKind arg, BlockIndex block_index)
-> ErrorOr<Success>;
// Verifies that `operand_id`, used by `user_id` in `block_index`, is either
// constant or dominated by an evaluation.
auto VerifyOperand(InstId user_id, InstId operand_id, BlockIndex block_index)
-> ErrorOr<Success>;
// Records that `inst_id` is evaluated at the point currently being verified.
auto RecordEvaluated(InstId inst_id) -> void;
// Records that every instruction in `block_id`, if it has a value, is
// evaluated at the point currently being verified.
auto RecordEvaluatedBlock(InstBlockId block_id) -> void;
// Returns the instruction that `splice` splices in, or `InstId::None` if that
// can't be determined.
auto GetSplicedInstId(SpliceInst splice) const -> InstId;
const File& file_;
const InstSet& decl_insts_;
const Function& function_;
const DominatorTree& dom_tree_;
SpecificId specific_id_;
llvm::ArrayRef<InstBlockId> body_blocks_;
// The instructions whose evaluations dominate the point currently being
// verified, and the order in which they were added, so that they can be
// removed again when leaving a block.
//
// `evaluated_` is owned by the caller and shared between functions, because
// creating one costs a bit per instruction in the whole file. `Verify` leaves
// it empty.
InstSet& evaluated_;
llvm::SmallVector<InstId> evaluated_order_;
};
auto DominanceVerifier::Verify() -> ErrorOr<Success> {
CARBON_CHECK(!body_blocks_.empty());
// Parameters and other instructions from the function declaration are
// evaluated before the body begins, so they dominate the whole body.
RecordEvaluatedBlock(function_.call_params_id);
RecordEvaluatedBlock(function_.call_param_patterns_id);
RecordEvaluatedBlock(function_.pattern_block_id);
RecordEvaluated(function_.self_param_id);
RecordEvaluated(function_.return_form_inst_id);
RecordEvaluated(function_.return_pattern_id);
for (InstId decl_id :
{function_.definition_id, function_.first_owning_decl_id,
function_.non_owning_decl_id}) {
if (!decl_id.has_value()) {
continue;
}
if (auto decl = file_.insts().TryGetAs<FunctionDecl>(decl_id)) {
RecordEvaluatedBlock(decl->decl_block_id);
}
}
auto result = VerifyBlocks();
// `evaluated_` is shared with the verification of other functions, so put it
// back the way it was found. Every instruction this added to it is in
// `evaluated_order_`, including if `VerifyBlocks` stopped at an error.
for (InstId inst_id : evaluated_order_) {
evaluated_.Erase(inst_id);
}
evaluated_order_.clear();
return result;
}
auto DominatorTreeBuilder::BuildControlFlowGraph() -> ErrorOr<Success> {
for (auto [i, block] : llvm::enumerate(body_blocks_)) {
block_indexes_.Insert(block, BlockIndex(i));
}
successors_.resize(num_blocks());
predecessors_.resize(num_blocks());
for (auto [i, block] : llvm::enumerate(body_blocks_)) {
BlockIndex from(i);
for (InstId inst_id : file_.inst_blocks().Get(block)) {
InstBlockId target_id = GetBranchTargetId(file_.insts().Get(inst_id));
if (!target_id.has_value()) {
continue;
}
BlockIndex* to = block_indexes_[target_id];
if (!to) {
return ErrorBuilder()
<< "Branch in block " << block << " targets block " << target_id
<< " which is not in function body";
}
// A block that branches to the same target more than once produces a
// duplicate edge. That's harmless: the post-order walk skips blocks it
// has already visited, and intersecting the dominators of the same
// predecessor twice gives the same result. Removing duplicates here
// would instead be quadratic in a block's number of successors.
successors_[from.index].push_back(*to);
predecessors_[to->index].push_back(from);
}
}
return Success();
}
auto DominatorTreeBuilder::BuildPostOrder(
llvm::SmallBitVector& visited,
llvm::SmallVectorImpl<BlockIndex>& post_order) -> void {
// The blocks whose successors are still being visited, each paired with the
// number of its successors that have been visited so far. A block is appended
// to `post_order` once all of its successors have been visited.
llvm::SmallVector<std::pair<BlockIndex, int>, 30> stack;
visited.set(EntryBlockIndex.index);
stack.push_back({EntryBlockIndex, 0});
while (!stack.empty()) {
auto [block_index, num_visited] = stack.back();
const auto& successors = successors_[block_index.index];
if (num_visited == static_cast<int>(successors.size())) {
post_order.push_back(block_index);
stack.pop_back();
continue;
}
stack.back().second = num_visited + 1;
BlockIndex successor = successors[num_visited];
if (!visited.test(successor.index)) {
visited.set(successor.index);
stack.push_back({successor, 0});
}
}
}
auto DominatorTreeBuilder::Build() -> ErrorOr<DominatorTree> {
CARBON_CHECK(!body_blocks_.empty());
CARBON_RETURN_IF_ERROR(BuildControlFlowGraph());
// Order the blocks so that, apart from loop back edges, every block precedes
// its successors.
llvm::SmallBitVector visited(num_blocks());
llvm::SmallVector<BlockIndex> reverse_post_order;
reverse_post_order.reserve(num_blocks());
BuildPostOrder(visited, reverse_post_order);
std::reverse(reverse_post_order.begin(), reverse_post_order.end());
for (auto [i, block] : llvm::enumerate(body_blocks_)) {
if (!visited.test(i)) {
return ErrorBuilder()
<< "Block " << block << " in function " << function_.name_id
<< " is unreachable from entry block";
}
}
llvm::SmallVector<int> order(num_blocks(), -1);
for (auto i : llvm::seq(num_blocks())) {
order[reverse_post_order[i].index] = i;
}
// Find the immediate dominator of each block using the algorithm from "A
// Simple, Fast Dominance Algorithm" by Cooper, Harvey, and Kennedy:
// repeatedly sweep the blocks in reverse post-order, intersecting the
// dominators of each block's predecessors, until the result stops changing.
llvm::SmallVector<BlockIndex> idom(num_blocks(), BlockIndex::None);
idom[EntryBlockIndex.index] = EntryBlockIndex;
// Returns the closest block that dominates both `lhs` and `rhs`, by walking
// both up the dominator tree built so far until they meet.
auto intersect = [&](BlockIndex lhs, BlockIndex rhs) -> BlockIndex {
while (lhs != rhs) {
while (order[lhs.index] > order[rhs.index]) {
lhs = idom[lhs.index];
}
while (order[rhs.index] > order[lhs.index]) {
rhs = idom[rhs.index];
}
}
return lhs;
};
for (bool changed = true; changed;) {
changed = false;
// The entry block is its own immediate dominator, so start after it.
for (BlockIndex block_index : llvm::drop_begin(reverse_post_order)) {
BlockIndex new_idom = BlockIndex::None;
for (BlockIndex predecessor : predecessors_[block_index.index]) {
if (!idom[predecessor.index].has_value()) {
continue;
}
new_idom = new_idom.has_value() ? intersect(predecessor, new_idom)
: predecessor;
}
if (new_idom.has_value() && idom[block_index.index] != new_idom) {
idom[block_index.index] = new_idom;
changed = true;
}
}
}
DominatorTree dom_children(num_blocks());
for (BlockIndex block_index : llvm::drop_begin(reverse_post_order)) {
// Every block other than the entry block is reachable, and so is dominated
// by the predecessor it's reached through.
CARBON_CHECK(idom[block_index.index].has_value(), "No dominator for {0}",
body_blocks_[block_index.index]);
dom_children[idom[block_index.index].index].push_back(block_index);
}
return dom_children;
}
auto DominanceVerifier::VerifyBlocks() -> ErrorOr<Success> {
llvm::SmallVector<WalkStep, 30> worklist = {
WalkStep::EnterBlock(EntryBlockIndex)};
while (!worklist.empty()) {
auto [block_index, scope_start] = worklist.pop_back_val();
if (!block_index.has_value()) {
for (InstId inst_id : llvm::drop_begin(evaluated_order_, scope_start)) {
evaluated_.Erase(inst_id);
}
evaluated_order_.truncate(scope_start);
continue;
}
// Evaluations in this block dominate the rest of this block and the blocks
// below it in the dominator tree, but nothing else. This step is beneath
// this block's children on the worklist, so it runs once they're done.
worklist.push_back(WalkStep::LeaveBlock(evaluated_order_.size()));
for (InstId inst_id :
file_.inst_blocks().Get(body_blocks_[block_index.index])) {
CARBON_RETURN_IF_ERROR(VerifyAndRecordInst(inst_id, block_index));
}
for (BlockIndex child : dom_tree_[block_index.index]) {
worklist.push_back(WalkStep::EnterBlock(child));
}
}
return Success();
}
auto DominanceVerifier::VerifyAndRecordInst(InstId root_inst_id,
BlockIndex block_index)
-> ErrorOr<Success> {
struct Step {
InstId inst_id;
// Whether to finish a `SpliceBlock` after its block has been evaluated.
bool finish_splice_block = false;
};
llvm::SmallVector<Step> worklist = {{.inst_id = root_inst_id}};
while (!worklist.empty()) {
auto [inst_id, finish_splice_block] = worklist.pop_back_val();
if (finish_splice_block) {
auto splice_block = file_.insts().GetAs<SpliceBlock>(inst_id);
CARBON_RETURN_IF_ERROR(
VerifyOperand(inst_id, splice_block.result_id, block_index));
RecordEvaluated(inst_id);
continue;
}
Inst inst = file_.insts().Get(inst_id);
// A `SpliceBlock` evaluates the instructions in its block, and then
// produces the value of one of them.
if (auto splice_block = inst.TryAs<SpliceBlock>()) {
worklist.push_back({.inst_id = inst_id, .finish_splice_block = true});
if (splice_block->block_id.has_value()) {
for (InstId spliced_id :
llvm::reverse(file_.inst_blocks().Get(splice_block->block_id))) {
worklist.push_back({.inst_id = spliced_id});
}
}
continue;
}
// A `SpliceInst` evaluates the instruction that its operand names.
if (auto splice = inst.TryAs<SpliceInst>()) {
CARBON_RETURN_IF_ERROR(
VerifyOperand(inst_id, splice->inst_id, block_index));
RecordEvaluated(inst_id);
if (InstId spliced_id = GetSplicedInstId(*splice);
spliced_id.has_value()) {
worklist.push_back({.inst_id = spliced_id});
}
continue;
}
CARBON_RETURN_IF_ERROR(
VerifyArg(inst_id, inst.arg0_and_kind(), block_index));
CARBON_RETURN_IF_ERROR(
VerifyArg(inst_id, inst.arg1_and_kind(), block_index));
RecordEvaluated(inst_id);
}
return Success();
}
auto DominanceVerifier::VerifyArg(InstId user_id, IdAndKind arg,
BlockIndex block_index) -> ErrorOr<Success> {
CARBON_KIND_SWITCH(arg) {
// These operand kinds name the value produced by another instruction, so
// that instruction's evaluation must dominate this use.
case CARBON_KIND(InstId inst_id): {
return VerifyOperand(user_id, inst_id, block_index);
}
case CARBON_KIND(DestInstId inst_id): {
return VerifyOperand(user_id, inst_id, block_index);
}
// A `TypeInstId` always names a constant of type `type`, so this check
// should always pass, but checking it means we notice if that stops being
// true.
case CARBON_KIND(TypeInstId inst_id): {
return VerifyOperand(user_id, inst_id, block_index);
}
case CARBON_KIND(InstBlockId block_id): {
if (block_id.has_value()) {
for (InstId operand_id : file_.inst_blocks().Get(block_id)) {
CARBON_RETURN_IF_ERROR(
VerifyOperand(user_id, operand_id, block_index));
}
}
return Success();
}
default: {
// Every other operand kind either doesn't name an instruction at all, or
// names one in a way that doesn't require dominance:
//
// - `MetaInstId` and `MetaInstBlockId` name the identity of an
// instruction rather than its value.
// - `AbsoluteInstId` and `AbsoluteInstBlockId` name instructions that
// are typically in a different entity.
// - `LabelId` names another block in this function's control flow.
// - `DeclInstBlockId` names a declaration rather than a computation.
return Success();
}
}
}
auto DominanceVerifier::VerifyOperand(InstId user_id, InstId operand_id,
BlockIndex block_index)
-> ErrorOr<Success> {
if (!operand_id.has_value() || operand_id == ErrorInst::InstId) {
return Success();
}
// A constant isn't evaluated in the function body, so can be used anywhere.
//
// TODO: Use `GetConstantValueInSpecific` here, so that an instruction that is
// only constant in this specific is also exempt. That currently crashes,
// because a function body can name an instruction whose constant value is
// attached to an enclosing generic rather than to this function's generic,
// which `GetConstantInSpecific` rejects. Lowering should hit the same crash;
// see `FunctionContext::LowerInst`.
if (file_.constant_values().Get(operand_id).is_constant()) {
return Success();
}
if (evaluated_.Contains(operand_id)) {
return Success();
}
// TODO: Remove this allowlist; see `CollectDeclInsts`.
if (decl_insts_.Contains(operand_id)) {
return Success();
}
Inst operand = file_.insts().Get(operand_id);
// TODO: A non-constant import isn't evaluated in the importing file at all,
// so this is a real violation: `__global_init` can contain a `name_ref` to an
// `import_ref` for a non-constant imported variable. Remove this exemption
// and diagnose such uses once imported variables have a value model.
if (operand.Is<AnyImportRef>()) {
return Success();
}
return ErrorBuilder()
<< "Operand " << operand_id << " (" << operand.kind().ir_name()
<< ") used by instruction " << user_id << " ("
<< file_.insts().Get(user_id).kind().ir_name() << ") in block "
<< body_blocks_[block_index.index] << " of function "
<< function_.name_id
<< " is not dominated by any evaluation and is not constant";
}
auto DominanceVerifier::RecordEvaluated(InstId inst_id) -> void {
if (!inst_id.has_value()) {
return;
}
// An instruction can be evaluated more than once, for example in two blocks
// that don't dominate each other. Only the first evaluation in scope needs to
// be undone.
if (evaluated_.Insert(inst_id)) {
evaluated_order_.push_back(inst_id);
}
}
auto DominanceVerifier::RecordEvaluatedBlock(InstBlockId block_id) -> void {
if (!block_id.has_value()) {
return;
}
for (InstId inst_id : file_.inst_blocks().Get(block_id)) {
RecordEvaluated(inst_id);
}
}
auto DominanceVerifier::GetSplicedInstId(SpliceInst splice) const -> InstId {
ConstantId const_id =
GetConstantValueInSpecific(file_, specific_id_, splice.inst_id);
if (!const_id.is_constant()) {
return InstId::None;
}
InstId const_inst_id = file_.constant_values().GetInstIdIfValid(const_id);
if (!const_inst_id.has_value()) {
return InstId::None;
}
if (auto inst_value = file_.insts().TryGetAs<InstValue>(const_inst_id)) {
return inst_value->inst_id;
}
return InstId::None;
}
} // namespace
auto VerifyDominance(const File& file) -> ErrorOr<Success> {
// Invariants don't necessarily hold for invalid IR.
if (file.has_errors()) {
return Success();
}
InstSet decl_insts(file);
CollectDeclInsts(file, decl_insts);
SpecificsByGeneric specifics = CollectSpecifics(file);
// Shared by the verification of every function and specific, because creating
// one costs a bit per instruction in the file. Each `Verify` call leaves it
// empty for the next one.
InstSet evaluated(file);
for (const Function& function : file.functions().values()) {
if (function.body_block_ids.empty()) {
continue;
}
CARBON_ASSIGN_OR_RETURN(DominatorTree dom_tree,
DominatorTreeBuilder(file, function).Build());
// Verify the body in the general, unspecialized context.
CARBON_RETURN_IF_ERROR(DominanceVerifier(file, decl_insts, function,
dom_tree, evaluated,
SpecificId::None)
.Verify());
// For a generic function, also verify the body as it will be evaluated in
// each of its specifics, in which spliced instructions can be resolved.
// TODO: Only re-verify the spliced instructions, rather than all the
// instructions in the function, for each specific.
for (SpecificId specific_id : specifics.Get(function.generic_id)) {
CARBON_RETURN_IF_ERROR(DominanceVerifier(file, decl_insts, function,
dom_tree, evaluated, specific_id)
.Verify());
}
}
return Success();
}
} // namespace Carbon::SemIR
+26
View File
@@ -0,0 +1,26 @@
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#ifndef CARBON_TOOLCHAIN_SEM_IR_DOMINANCE_H_
#define CARBON_TOOLCHAIN_SEM_IR_DOMINANCE_H_
#include "common/error.h"
namespace Carbon::SemIR {
class File;
// Verifies that every use of an instruction's value within a function body is
// dominated by an evaluation of that instruction, or that the instruction has a
// constant value. This is checked for every function definition in `file`, and
// for each resolved specific of a generic function, where spliced instructions
// can be resolved.
//
// Nothing is checked if `file` has errors, because these invariants don't
// necessarily hold for invalid IR.
auto VerifyDominance(const File& file) -> ErrorOr<Success>;
} // namespace Carbon::SemIR
#endif // CARBON_TOOLCHAIN_SEM_IR_DOMINANCE_H_
+283
View File
@@ -0,0 +1,283 @@
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include <benchmark/benchmark.h>
#include <algorithm>
#include "common/check.h"
#include "common/error.h"
#include "llvm/ADT/ArrayRef.h"
#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/Sequence.h"
#include "llvm/ADT/SmallVector.h"
#include "toolchain/sem_ir/dominance.h"
#include "toolchain/sem_ir/dominance_test_helpers.h"
#include "toolchain/sem_ir/ids.h"
namespace Carbon::SemIR {
namespace {
// The number of value uses in each generated block, in addition to its
// terminators. Real blocks contain more instructions than terminators, so this
// spreads the verifier's per-block costs over several instructions, as they
// would be in practice.
constexpr int InstsPerBlock = 8;
// Builds a file containing synthetic function bodies to verify.
class FileBuilder : public DominanceTestFile {
public:
// Fills the entry block of a function body. It evaluates the value that the
// rest of the body uses, so that every use in the body is dominated, and
// branches to `next_id`.
auto FillEntryBlock(InstBlockId block_id, InstBlockId next_id) -> void {
SetBlock(block_id, {value_id_, AddBranch(next_id)});
}
// Fills a non-entry block with `num_uses` uses of the value evaluated in the
// entry block, followed by the block's terminators.
auto FillBlock(InstBlockId block_id, llvm::ArrayRef<InstId> terminators,
int num_uses = InstsPerBlock) -> void {
llvm::SmallVector<InstId> insts;
insts.reserve(num_uses);
for (auto _ : llvm::seq(num_uses)) {
insts.push_back(AddUse(value_id_));
}
SetBlock(block_id, insts, terminators);
}
private:
// The non-constant value that every generated block uses.
InstId value_id_ = AddValue();
};
// Builds a function body with a given control flow shape and approximate
// number of blocks, and returns its blocks, entry block first.
using BuildBodyFn = auto (*)(FileBuilder& file, int num_blocks)
-> llvm::SmallVector<InstBlockId>;
// A single large block:
//
// entry -> body
//
// This is the baseline: it has the same number of instructions as the other
// shapes of the same size, but essentially no control flow, so it measures the
// cost per instruction rather than per block.
auto BuildStraightLine(FileBuilder& file, int num_blocks)
-> llvm::SmallVector<InstBlockId> {
auto entry_id = file.AddBlock();
auto body_id = file.AddBlock();
file.FillEntryBlock(entry_id, body_id);
file.FillBlock(body_id, {file.AddReturn()},
/*num_uses=*/num_blocks * InstsPerBlock);
return {entry_id, body_id};
}
// A chain of blocks:
//
// entry -> b1 -> b2 -> ... -> bn
//
// The dominator tree is a single deep path, so the set of instructions whose
// evaluations dominate the point being verified grows to the size of the body.
auto BuildChain(FileBuilder& file, int num_blocks)
-> llvm::SmallVector<InstBlockId> {
int size = std::max(num_blocks, 2);
llvm::SmallVector<InstBlockId> blocks;
blocks.reserve(size);
for (auto _ : llvm::seq(size)) {
blocks.push_back(file.AddBlock());
}
file.FillEntryBlock(blocks[0], blocks[1]);
for (auto i : llvm::seq(1, size)) {
file.FillBlock(blocks[i], {i + 1 == size ? file.AddReturn()
: file.AddBranch(blocks[i + 1])});
}
return blocks;
}
// A chain of diamonds, as an `if` in a loop body would produce:
//
// entry -> head0 -> then0 -> head1 -> then1 -> ... -> headn
// \ -> else0 -/ \ -> else1 -/
//
// Each join block has two predecessors, so this exercises the dominator
// computation's intersection step, and the dominator tree is both deep and
// branching.
auto BuildDiamonds(FileBuilder& file, int num_blocks)
-> llvm::SmallVector<InstBlockId> {
int num_diamonds = std::max((num_blocks - 2) / 3, 1);
llvm::SmallVector<InstBlockId> blocks = {file.AddBlock()};
llvm::SmallVector<InstBlockId> heads;
for (auto _ : llvm::seq(num_diamonds + 1)) {
heads.push_back(file.AddBlock());
blocks.push_back(heads.back());
}
file.FillEntryBlock(blocks[0], heads[0]);
for (auto i : llvm::seq(num_diamonds)) {
auto then_id = file.AddBlock();
auto else_id = file.AddBlock();
blocks.push_back(then_id);
blocks.push_back(else_id);
file.FillBlock(heads[i],
{file.AddBranchIf(then_id), file.AddBranch(else_id)});
file.FillBlock(then_id, {file.AddBranch(heads[i + 1])});
file.FillBlock(else_id, {file.AddBranch(heads[i + 1])});
}
file.FillBlock(heads[num_diamonds], {file.AddReturn()});
return blocks;
}
// A wide fan-out and fan-in, as a `match` would produce:
//
// entry -> head -> arm1 -> exit
// \ -> ... -/
// \ -> armn -/
//
// The join block has a predecessor per arm, so this is the worst case for the
// parts of the verifier that are quadratic in a block's number of edges.
auto BuildFanOutFanIn(FileBuilder& file, int num_blocks)
-> llvm::SmallVector<InstBlockId> {
int num_arms = std::max(num_blocks - 3, 1);
auto entry_id = file.AddBlock();
auto head_id = file.AddBlock();
auto exit_id = file.AddBlock();
llvm::SmallVector<InstBlockId> blocks = {entry_id, head_id, exit_id};
llvm::SmallVector<InstId> head_terminators;
head_terminators.reserve(num_arms);
for (auto i : llvm::seq(num_arms)) {
auto arm_id = file.AddBlock();
blocks.push_back(arm_id);
// The last arm is reached unconditionally, as the `else` of the last
// condition.
head_terminators.push_back(i + 1 == num_arms ? file.AddBranch(arm_id)
: file.AddBranchIf(arm_id));
file.FillBlock(arm_id, {file.AddBranch(exit_id)});
}
file.FillEntryBlock(entry_id, head_id);
file.FillBlock(head_id, head_terminators);
file.FillBlock(exit_id, {file.AddReturn()});
return blocks;
}
// Nested loops:
//
// entry -> header1 -> body1 -> header2 -> body2 -> ... -> (back edge)
// \ -> exit1 <- exit2 <- ...
//
// Each loop's back edge is a predecessor that is visited after the block it
// targets, so the dominator computation needs repeated sweeps to converge.
auto BuildNestedLoops(FileBuilder& file, int num_blocks)
-> llvm::SmallVector<InstBlockId> {
int num_loops = std::max((num_blocks - 1) / 3, 1);
auto entry_id = file.AddBlock();
llvm::SmallVector<InstBlockId> blocks = {entry_id};
llvm::SmallVector<InstBlockId> headers;
llvm::SmallVector<InstBlockId> bodies;
llvm::SmallVector<InstBlockId> exits;
for (auto _ : llvm::seq(num_loops)) {
headers.push_back(file.AddBlock());
bodies.push_back(file.AddBlock());
exits.push_back(file.AddBlock());
blocks.append({headers.back(), bodies.back(), exits.back()});
}
file.FillEntryBlock(entry_id, headers[0]);
for (auto [i, header, body, exit] : llvm::enumerate(headers, bodies, exits)) {
file.FillBlock(header, {file.AddBranchIf(body), file.AddBranch(exit)});
// The innermost body closes its own loop; every other body enters the next
// loop, whose exit branches back to this header.
file.FillBlock(body, {file.AddBranch(static_cast<int>(i + 1) == num_loops
? header
: headers[i + 1])});
file.FillBlock(
exit, {i == 0 ? file.AddReturn() : file.AddBranch(headers[i - 1])});
}
return blocks;
}
// Verifies `file` repeatedly, and reports the rate at which its instructions
// are verified.
auto RunBenchmark(benchmark::State& state, FileBuilder& file) -> void {
for (auto _ : state) {
ErrorOr<Success> result = VerifyDominance(file.file());
CARBON_CHECK(result.ok(), "{0}", result.error().message());
}
state.counters["insts_per_second"] =
benchmark::Counter(static_cast<double>(file.file().insts().size()),
benchmark::Counter::kIsIterationInvariantRate);
}
// The body sizes to benchmark, in blocks. The largest is far bigger than any
// realistic function body, so that superlinear behavior shows up as a falling
// instruction rate.
auto BodySizes(benchmark::Benchmark* bench) -> void {
bench->RangeMultiplier(8)->Range(8, 32768);
}
// Verifies one function whose body has the shape built by `Build`.
template <BuildBodyFn Build>
auto BM_VerifyDominance(benchmark::State& state) -> void {
FileBuilder file;
file.AddFunction(Build(file, state.range(0)));
RunBenchmark(state, file);
}
BENCHMARK(BM_VerifyDominance<BuildStraightLine>)
->Name("BM_VerifyDominance/StraightLine")
->Apply(BodySizes);
BENCHMARK(BM_VerifyDominance<BuildChain>)
->Name("BM_VerifyDominance/Chain")
->Apply(BodySizes);
BENCHMARK(BM_VerifyDominance<BuildDiamonds>)
->Name("BM_VerifyDominance/Diamonds")
->Apply(BodySizes);
BENCHMARK(BM_VerifyDominance<BuildNestedLoops>)
->Name("BM_VerifyDominance/NestedLoops")
->Apply(BodySizes);
// This shape has a block with a successor for each arm and a block with a
// predecessor for each arm, so it's the shape to watch for work that's
// quadratic in a block's number of edges.
BENCHMARK(BM_VerifyDominance<BuildFanOutFanIn>)
->Name("BM_VerifyDominance/FanOutFanIn")
->Apply(BodySizes);
// Verifies many small functions, which is the shape of a real file: this
// measures the verifier's per-function costs rather than its scaling within a
// function body.
auto BM_VerifyDominanceManyFunctions(benchmark::State& state) -> void {
FileBuilder file;
for (auto _ : llvm::seq(state.range(0))) {
file.AddFunction(BuildDiamonds(file, /*num_blocks=*/5));
}
RunBenchmark(state, file);
}
BENCHMARK(BM_VerifyDominanceManyFunctions)
->Name("BM_VerifyDominance/ManyFunctions")
->Apply(BodySizes);
// Verifies many small generic functions, each with one resolved specific.
// Every generic function's body is verified once per specific of its generic,
// so this watches for work that's quadratic in the number of generics in the
// file rather than linear in the number of specifics.
auto BM_VerifyDominanceManyGenericFunctions(benchmark::State& state) -> void {
FileBuilder file;
for (auto _ : llvm::seq(state.range(0))) {
file.AddFunction(BuildDiamonds(file, /*num_blocks=*/5), file.AddGeneric());
}
RunBenchmark(state, file);
}
BENCHMARK(BM_VerifyDominanceManyGenericFunctions)
->Name("BM_VerifyDominance/ManyGenericFunctions")
->Apply(BodySizes);
} // namespace
} // namespace Carbon::SemIR
+482
View File
@@ -0,0 +1,482 @@
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include "toolchain/sem_ir/dominance.h"
#include <gmock/gmock.h>
#include <gtest/gtest.h>
#include <string>
#include "llvm/ADT/ArrayRef.h"
#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/Sequence.h"
#include "llvm/ADT/SmallVector.h"
#include "toolchain/parse/node_ids.h"
#include "toolchain/sem_ir/dominance_test_helpers.h"
#include "toolchain/sem_ir/ids.h"
#include "toolchain/sem_ir/typed_insts.h"
namespace Carbon::SemIR {
namespace {
using ::testing::HasSubstr;
using ::testing::IsEmpty;
class DominanceTest : public ::testing::Test, public DominanceTestFile {
public:
// Runs the dominance check, and returns the error message it produced, or an
// empty string if it succeeded.
auto Verify() -> std::string {
ErrorOr<Success> result = VerifyDominance(file());
return result.ok() ? "" : result.error().message();
}
// Adds an instruction that refers to `inst_id` as a `MetaInstId`, which names
// the instruction rather than its value.
auto AddMetaUse(InstId inst_id) -> InstId {
return AddNonConstInst(AccessMemberAction{.type_id = TypeType::TypeId,
.base_id = MetaInstId(inst_id),
.name_id = NameId(0)});
}
// Adds an instruction whose constant value is an `inst_value` naming
// `target_id`. This is the form that the operand of a `splice_inst` takes.
auto AddInstValue(InstId target_id) -> InstId {
auto value_id = AddInst(InstValue{.type_id = TypeType::TypeId,
.inst_id = MetaInstId(target_id)});
file().constant_values().Set(value_id,
ConstantId::ForConcreteConstant(value_id));
auto use_id = AddInst(InstValue{.type_id = TypeType::TypeId,
.inst_id = MetaInstId(target_id)});
file().constant_values().Set(use_id,
ConstantId::ForConcreteConstant(value_id));
return use_id;
}
};
TEST_F(DominanceTest, FunctionWithNoBody) {
AddFunction();
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceTest, StraightLineUseAfterEvaluation) {
auto value_id = AddValue();
auto use_id = AddUse(value_id);
AddFunction({AddBlock({value_id, use_id, AddReturn()})});
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceTest, StraightLineUseBeforeEvaluation) {
auto value_id = AddValue();
auto use_id = AddUse(value_id);
AddFunction({AddBlock({use_id, value_id, AddReturn()})});
EXPECT_THAT(Verify(), HasSubstr("not dominated by any evaluation"));
}
TEST_F(DominanceTest, NeverEvaluatedValue) {
// A non-constant value that is not evaluated anywhere in the body, as would
// happen for a non-constant global or import, doesn't dominate its uses.
auto global_id = AddValue();
auto use_id = AddUse(global_id);
AddFunction({AddBlock({use_id, AddReturn()})});
EXPECT_THAT(Verify(),
HasSubstr("not dominated by any evaluation and is not constant"));
}
TEST_F(DominanceTest, ConstantIsExempt) {
auto constant_id = AddConstant();
auto use_id = AddUse(constant_id);
AddFunction({AddBlock({use_id, AddReturn()})});
EXPECT_THAT(Verify(), IsEmpty());
}
// The following three tests pin the allowlists for pre-existing dominance
// violations. Each of them fails if the corresponding allowlist is removed, so
// they should be removed, and the violations diagnosed, together with it.
// See `CollectDeclInsts` and `DominanceVerifier::VerifyOperand`.
TEST_F(DominanceTest, FileScopeInstIsAllowlisted) {
// A file-scope instruction is evaluated in `__global_init`, if at all, so it
// doesn't dominate uses in any other function.
auto global_id = AddValue();
file().set_top_inst_block_id(AddBlock({global_id}));
auto use_id = AddUse(global_id);
AddFunction({AddBlock({use_id, AddReturn()})});
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceTest, ClassBodyInstIsAllowlisted) {
// A `let` in a class body produces a `wrapper_binding` that isn't evaluated
// in any function, but `A.x` can name it from one. This mirrors the
// `public_global_access` case in
// `check/testdata/class/access/access_modifiers.carbon`:
//
// class A { let x: i32 = 5; }
// let x: i32 = A.x;
auto binding_id =
AddNonConstInst(WrapperBinding{.type_id = TypeType::TypeId,
.entity_name_id = EntityNameId::None,
.value_id = AddConstant()});
auto name_id = file().identifiers().Add("A");
file().classes().Add({{.name_id = NameId::ForIdentifier(name_id),
.parent_scope_id = NameScopeId::Package,
.generic_id = GenericId::None,
.first_param_node_id = Parse::NodeId::None,
.last_param_node_id = Parse::NodeId::None,
.pattern_block_id = InstBlockId::Empty,
.implicit_param_patterns_id = InstBlockId::None,
.param_patterns_id = InstBlockId::Empty,
.is_extern = false,
.extern_library_id = LibraryNameId::None,
.non_owning_decl_id = InstId::None,
.first_owning_decl_id = InstId::None},
{.self_type_id = TypeType::TypeId,
.inheritance_kind = Class::Final,
.body_block_id = AddBlock({binding_id})}});
auto use_id = AddUse(binding_id);
AddFunction({AddBlock({use_id, AddReturn()})});
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceTest, ImportRefIsAllowlisted) {
// A non-constant import isn't evaluated in the importing file at all.
auto import_id =
AddNonConstInst(ImportRefLoaded{.type_id = TypeType::TypeId,
.import_ir_inst_id = ImportIRInstId::None,
.entity_name_id = EntityNameId::None});
auto use_id = AddUse(import_id);
AddFunction({AddBlock({use_id, AddReturn()})});
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceTest, MetaInstIdOperandIsExempt) {
// A `MetaInstId` names the identity of an instruction rather than its value,
// so it needn't be dominated even though it's evaluated later.
auto value_id = AddValue();
auto use_id = AddMetaUse(value_id);
AddFunction({AddBlock({use_id, value_id, AddReturn()})});
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceTest, ErroneousFileIsNotChecked) {
auto value_id = AddValue();
auto use_id = AddUse(value_id);
AddFunction({AddBlock({use_id, value_id, AddReturn()})});
file().set_has_errors(true);
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceTest, FileVerifyChecksDominance) {
// `File::Verify` should run the dominance check as well as its other checks.
auto value_id = AddValue();
auto use_id = AddUse(value_id);
AddFunction({AddBlock({use_id, value_id, AddReturn()})});
auto result = file().Verify();
ASSERT_FALSE(result.ok());
EXPECT_THAT(result.error().message(),
HasSubstr("not dominated by any evaluation"));
}
// A fixture for tests over a diamond:
//
// entry
// / \
// then else
// \ /
// exit
class DominanceDiamondTest : public DominanceTest {
public:
// Builds the diamond, prefixing each block with the given instructions.
auto BuildDiamond(llvm::ArrayRef<InstId> entry_insts,
llvm::ArrayRef<InstId> then_insts,
llvm::ArrayRef<InstId> else_insts,
llvm::ArrayRef<InstId> exit_insts) -> void {
auto entry_id = AddBlock();
auto then_id = AddBlock();
auto else_id = AddBlock();
auto exit_id = AddBlock();
SetBlock(entry_id, entry_insts, {AddBranchIf(then_id), AddBranch(else_id)});
SetBlock(then_id, then_insts, {AddBranch(exit_id)});
SetBlock(else_id, else_insts, {AddBranch(exit_id)});
SetBlock(exit_id, exit_insts, {AddReturn()});
AddFunction({entry_id, then_id, else_id, exit_id});
}
};
TEST_F(DominanceDiamondTest, EvaluationInEntryDominatesAllBranches) {
auto value_id = AddValue();
BuildDiamond(/*entry_insts=*/{value_id}, /*then_insts=*/{AddUse(value_id)},
/*else_insts=*/{AddUse(value_id)},
/*exit_insts=*/{AddUse(value_id)});
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceDiamondTest, EvaluationInOneBranchDoesNotDominateTheOther) {
auto value_id = AddValue();
BuildDiamond(/*entry_insts=*/{}, /*then_insts=*/{value_id},
/*else_insts=*/{AddUse(value_id)}, /*exit_insts=*/{});
EXPECT_THAT(Verify(), HasSubstr("not dominated by any evaluation"));
}
TEST_F(DominanceDiamondTest, EvaluationInOneBranchDoesNotDominateJoin) {
auto value_id = AddValue();
BuildDiamond(/*entry_insts=*/{}, /*then_insts=*/{value_id},
/*else_insts=*/{}, /*exit_insts=*/{AddUse(value_id)});
EXPECT_THAT(Verify(), HasSubstr("not dominated by any evaluation"));
}
TEST_F(DominanceDiamondTest, EvaluationInBothBranchesDoesNotDominateJoin) {
// The same instruction can be evaluated in more than one block; a use is
// dominated if it's dominated by any of the evaluations. Here neither
// evaluation dominates the join, so this is still an error.
auto value_id = AddValue();
BuildDiamond(/*entry_insts=*/{}, /*then_insts=*/{value_id},
/*else_insts=*/{value_id}, /*exit_insts=*/{AddUse(value_id)});
EXPECT_THAT(Verify(), HasSubstr("not dominated by any evaluation"));
}
TEST_F(DominanceDiamondTest, MultipleEvaluationsEachDominateTheirOwnUse) {
auto value_id = AddValue();
BuildDiamond(/*entry_insts=*/{},
/*then_insts=*/{value_id, AddUse(value_id)},
/*else_insts=*/{value_id, AddUse(value_id)},
/*exit_insts=*/{});
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceTest, UnreachableBlock) {
auto entry_id = AddBlock({AddReturn()});
auto unreachable_id = AddBlock({AddReturn()});
AddFunction({entry_id, unreachable_id});
EXPECT_THAT(Verify(), HasSubstr("is unreachable from entry block"));
}
TEST_F(DominanceTest, BranchOutsideFunctionBody) {
auto outside_id = AddBlock({AddReturn()});
auto entry_id = AddBlock({AddBranch(outside_id)});
AddFunction({entry_id});
EXPECT_THAT(Verify(), HasSubstr("which is not in function body"));
}
TEST_F(DominanceTest, RepeatedBranchToSameBlock) {
// A block can branch to the same block more than once, for example when both
// arms of an `if` are empty. That produces a duplicate control flow edge,
// which shouldn't disturb the dominance computation: the exit block is still
// dominated by the entry block.
auto entry_id = AddBlock();
auto exit_id = AddBlock();
auto value_id = AddValue();
SetBlock(entry_id, {value_id, AddBranchIf(exit_id), AddBranch(exit_id)});
SetBlock(exit_id, {AddUse(value_id), AddReturn()});
AddFunction({entry_id, exit_id});
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceTest, LongChainOfBlocks) {
// A function body long enough that walking it recursively would overflow the
// stack. Each block uses a value evaluated in the block before it, so the
// dominator tree is a single chain.
constexpr int NumBlocks = 100'000;
llvm::SmallVector<InstBlockId> block_ids;
block_ids.reserve(NumBlocks);
for (auto _ : llvm::seq(NumBlocks)) {
block_ids.push_back(AddBlock());
}
auto value_id = AddValue();
for (auto [i, block] : llvm::enumerate(block_ids)) {
SetBlock(block,
{i == 0 ? value_id : AddUse(value_id),
i + 1 == NumBlocks ? AddReturn() : AddBranch(block_ids[i + 1])});
}
AddFunction(block_ids);
EXPECT_THAT(Verify(), IsEmpty());
}
// A loop:
//
// entry -> header -> body -> header
// -> exit
class DominanceLoopTest : public DominanceTest {
public:
// Builds the loop, prefixing the header and body blocks with the given
// instructions.
auto BuildLoop(llvm::ArrayRef<InstId> header_insts,
llvm::ArrayRef<InstId> body_insts) -> void {
auto entry_id = AddBlock();
auto header_id = AddBlock();
auto body_id = AddBlock();
auto exit_id = AddBlock();
SetBlock(entry_id, {AddBranch(header_id)});
SetBlock(header_id, header_insts,
{AddBranchIf(body_id), AddBranch(exit_id)});
SetBlock(body_id, body_insts, {AddBranch(header_id)});
SetBlock(exit_id, {AddReturn()});
AddFunction({entry_id, header_id, body_id, exit_id});
}
};
TEST_F(DominanceLoopTest, HeaderEvaluationDominatesBody) {
auto value_id = AddValue();
BuildLoop(/*header_insts=*/{value_id}, /*body_insts=*/{AddUse(value_id)});
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceLoopTest, BodyEvaluationDoesNotDominateHeader) {
// The back edge means the body is executed before the header on some paths,
// but not on the path that first reaches the header.
auto value_id = AddValue();
BuildLoop(/*header_insts=*/{AddUse(value_id)}, /*body_insts=*/{value_id});
EXPECT_THAT(Verify(), HasSubstr("not dominated by any evaluation"));
}
TEST_F(DominanceTest, SpliceBlockEvaluatesItsContents) {
auto generic_id = AddGeneric();
auto value_id = AddValue();
// A spliced block that evaluates a use of `value_id` and produces it.
auto inner_id = AddUse(value_id);
auto splice_block_id =
AddInst(SpliceBlock{.type_id = TypeType::TypeId,
.block_id = AbsoluteInstBlockId(AddBlock({inner_id})),
.result_id = inner_id});
auto action_id = AddInstValue(splice_block_id);
auto splice_id =
AddInst(SpliceInst{.type_id = TypeType::TypeId, .inst_id = action_id});
// The use after the splice sees the instructions the splice evaluated.
auto late_use_id = AddUse(inner_id);
AddFunction(
{AddBlock({value_id, action_id, splice_id, late_use_id, AddReturn()})},
generic_id);
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceTest, SpliceBlockUsingLaterValue) {
auto generic_id = AddGeneric();
// A spliced block that uses a value evaluated after the splice.
auto value_id = AddValue();
auto inner_id = AddUse(value_id);
auto splice_block_id =
AddInst(SpliceBlock{.type_id = TypeType::TypeId,
.block_id = AbsoluteInstBlockId(AddBlock({inner_id})),
.result_id = inner_id});
auto action_id = AddInstValue(splice_block_id);
auto splice_id =
AddInst(SpliceInst{.type_id = TypeType::TypeId, .inst_id = action_id});
AddFunction({AddBlock({action_id, splice_id, value_id, AddReturn()})},
generic_id);
EXPECT_THAT(Verify(), HasSubstr("not dominated by any evaluation"));
}
// A generic function whose body contains a `splice_inst` whose operand is a
// symbolic constant, so that the splice resolves to a different instruction
// when the body is checked as a generic than when it's checked in the generic's
// specific.
class DominanceSpecificSpliceTest : public DominanceTest {
protected:
// Builds the function. Its body evaluates `early_id`, then splices in an
// instruction, then evaluates `late_id`. The splice produces
// `generic_spliced_id` when the body is checked as a generic, and
// `specific_spliced_id` when it's checked in the specific.
auto BuildFunction(InstId early_id, InstId late_id, InstId generic_spliced_id,
InstId specific_spliced_id) -> void {
// In the specific, the splice's operand takes its value from the entry of
// the specific's value block that its symbolic constant indexes.
auto generic_id = AddGeneric(AddBlock({AddInstValue(specific_spliced_id)}));
// In the generic, it takes the value of the unattached form of that
// symbolic constant, which is the constant value of the instruction that
// defines the constant.
auto unattached_id =
AddInst(InstValue{.type_id = TypeType::TypeId,
.inst_id = MetaInstId(generic_spliced_id)});
file().constant_values().Set(
unattached_id, file().constant_values().AddSymbolicConstant(
{.inst_id = unattached_id,
.generic_id = GenericId::None,
.index = GenericInstIndex::None,
.dependence = ConstantDependence::Checked}));
auto action_id =
AddInst(InstValue{.type_id = TypeType::TypeId,
.inst_id = MetaInstId(generic_spliced_id)});
file().constant_values().Set(
action_id,
file().constant_values().AddSymbolicConstant(
{.inst_id = unattached_id,
.generic_id = generic_id,
.index = GenericInstIndex(GenericInstIndex::Declaration, 0),
.dependence = ConstantDependence::Checked}));
auto splice_id =
AddInst(SpliceInst{.type_id = TypeType::TypeId, .inst_id = action_id});
AddFunction(
{AddBlock({early_id, action_id, splice_id, late_id, AddReturn()})},
generic_id);
}
};
TEST_F(DominanceSpecificSpliceTest, SplicedUseInSpecificIsDominated) {
auto early_id = AddValue();
auto late_id = AddValue();
BuildFunction(early_id, late_id, /*generic_spliced_id=*/AddUse(early_id),
/*specific_spliced_id=*/AddUse(early_id));
EXPECT_THAT(Verify(), IsEmpty());
}
TEST_F(DominanceSpecificSpliceTest, SplicedUseInSpecificIsNotDominated) {
// The instruction that the specific splices in uses a value that isn't
// evaluated until after the splice. That's only visible when the body is
// checked in the specific, because the generic splices in a use of a value
// that is evaluated before it.
auto early_id = AddValue();
auto late_id = AddValue();
BuildFunction(early_id, late_id, /*generic_spliced_id=*/AddUse(early_id),
/*specific_spliced_id=*/AddUse(late_id));
EXPECT_THAT(Verify(), HasSubstr("not dominated by any evaluation"));
}
} // namespace
} // namespace Carbon::SemIR
+173
View File
@@ -0,0 +1,173 @@
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#ifndef CARBON_TOOLCHAIN_SEM_IR_DOMINANCE_TEST_HELPERS_H_
#define CARBON_TOOLCHAIN_SEM_IR_DOMINANCE_TEST_HELPERS_H_
#include <optional>
#include "llvm/ADT/ArrayRef.h"
#include "llvm/ADT/SmallVector.h"
#include "llvm/ADT/StringRef.h"
#include "toolchain/base/shared_value_stores.h"
#include "toolchain/parse/node_ids.h"
#include "toolchain/sem_ir/file.h"
#include "toolchain/sem_ir/function.h"
#include "toolchain/sem_ir/generic.h"
#include "toolchain/sem_ir/ids.h"
#include "toolchain/sem_ir/inst.h"
#include "toolchain/sem_ir/typed_insts.h"
namespace Carbon::SemIR {
// Builds a `SemIR::File` containing synthetic function bodies for dominance
// tests and benchmarks.
//
// The generated IR only has the properties that `VerifyDominance` looks at: it
// has no locations or real types, and instructions may be shared between
// functions, neither of which the verifier examines.
class DominanceTestFile {
public:
DominanceTestFile()
: file_(/*parse_tree=*/nullptr, CheckIRId(0),
/*packaging_decl=*/std::nullopt, value_stores_,
"dominance_test.carbon"),
cond_id_(AddConstant()) {}
auto file() -> File& { return file_; }
// Adds a block whose contents are filled in later by `SetBlock`.
auto AddBlock() -> InstBlockId {
return file_.inst_blocks().AddPlaceholder();
}
// Adds a block containing `insts` followed by `terminators`.
auto AddBlock(llvm::ArrayRef<InstId> insts,
llvm::ArrayRef<InstId> terminators = {}) -> InstBlockId {
auto block_id = AddBlock();
SetBlock(block_id, insts, terminators);
return block_id;
}
// Fills a block created by `AddBlock()` with `insts` followed by
// `terminators`.
auto SetBlock(InstBlockId block_id, llvm::ArrayRef<InstId> insts,
llvm::ArrayRef<InstId> terminators = {}) -> void {
if (terminators.empty()) {
file_.inst_blocks().ReplacePlaceholder(block_id, insts);
return;
}
llvm::SmallVector<InstId> all_insts;
all_insts.reserve(insts.size() + terminators.size());
all_insts.append(insts.begin(), insts.end());
all_insts.append(terminators.begin(), terminators.end());
file_.inst_blocks().ReplacePlaceholder(block_id, all_insts);
}
// Adds an instruction that is not in any block.
template <typename InstT>
auto AddInst(InstT inst) -> InstId {
return file_.insts().AddInNoBlock(LocIdAndInst::NoLoc(inst));
}
// Adds an instruction with no constant value, so that uses of it must be
// dominated by its evaluation.
template <typename InstT>
auto AddNonConstInst(InstT inst) -> InstId {
auto inst_id = AddInst(inst);
file_.constant_values().Set(inst_id, ConstantId::NotConstant);
return inst_id;
}
// Adds an instruction producing a non-constant value.
auto AddValue() -> InstId {
return AddNonConstInst(
BoolLiteral{.type_id = TypeType::TypeId, .value = BoolValue(false)});
}
// Adds an instruction producing a constant value.
auto AddConstant() -> InstId {
auto inst_id = AddInst(
BoolLiteral{.type_id = TypeType::TypeId, .value = BoolValue(true)});
file_.constant_values().Set(inst_id,
ConstantId::ForConcreteConstant(inst_id));
return inst_id;
}
// Adds an instruction that uses the value of `value_id`.
auto AddUse(InstId value_id) -> InstId {
return AddNonConstInst(
ValueAsRef{.type_id = TypeType::TypeId, .value_id = value_id});
}
auto AddReturn() -> InstId { return AddInst(Return{}); }
auto AddBranch(InstBlockId target_id) -> InstId {
return AddInst(Branch{.target_id = LabelId(target_id)});
}
auto AddBranchIf(InstBlockId target_id) -> InstId {
return AddInst(
BranchIf{.target_id = LabelId(target_id), .cond_id = cond_id_});
}
// Adds a generic, along with a specific for it that a function can be
// attached to. `VerifyDominance` verifies a generic function's body once for
// the generic itself and once for each resolved specific. The specific is
// resolved, with `value_block_id` as the value block for its declaration.
auto AddGeneric(InstBlockId value_block_id = InstBlockId::Empty)
-> GenericId {
auto decl_id =
AddInst(FunctionDecl{.type_id = TypeType::TypeId,
.function_id = FunctionId(0),
.decl_block_id = DeclInstBlockId::None});
auto generic_id =
file_.generics().Add(Generic{.decl_id = decl_id,
.bindings_id = InstBlockId::Empty,
.self_specific_id = SpecificId::None});
auto specific_id =
file_.specifics().GetOrAdd(generic_id, InstBlockId::Empty);
file_.specifics()
.Get(specific_id)
.SetValueBlock(GenericInstIndex::Declaration, value_block_id);
return generic_id;
}
auto AddFunction(llvm::ArrayRef<InstBlockId> body_block_ids = {},
GenericId generic_id = GenericId::None) -> FunctionId {
auto name_id = file_.identifiers().Add("F");
return file_.functions().Add(
{{.name_id = NameId::ForIdentifier(name_id),
.parent_scope_id = NameScopeId::Package,
.generic_id = generic_id,
.first_param_node_id = Parse::NodeId::None,
.last_param_node_id = Parse::NodeId::None,
.pattern_block_id = InstBlockId::Empty,
.implicit_param_patterns_id = InstBlockId::None,
.param_patterns_id = InstBlockId::Empty,
.is_extern = false,
.extern_library_id = LibraryNameId::None,
.non_owning_decl_id = InstId::None,
.first_owning_decl_id = InstId::None},
{.call_param_patterns_id = InstBlockId::Empty,
.call_params_id = InstBlockId::Empty,
.call_param_ranges = Function::CallParamIndexRanges::Empty,
.return_type_inst_id = TypeInstId::None,
.return_form_inst_id = InstId::None,
.return_pattern_id = InstId::None,
.body_block_ids = llvm::SmallVector<InstBlockId>(
body_block_ids.begin(), body_block_ids.end())}});
}
private:
SharedValueStores value_stores_;
File file_;
// The condition used by conditional branches. It's constant, so it needs no
// dominating evaluation.
InstId cond_id_;
};
} // namespace Carbon::SemIR
#endif // CARBON_TOOLCHAIN_SEM_IR_DOMINANCE_TEST_HELPERS_H_
+5 -5
View File
@@ -12,15 +12,15 @@
#include "clang/AST/Decl.h"
#include "clang/AST/Mangle.h"
#include "common/check.h"
#include "llvm/ADT/STLExtras.h"
#include "llvm/ADT/SmallVector.h"
#include "common/error.h"
#include "toolchain/base/block_value_store_impl.h"
#include "toolchain/base/kind_switch.h"
#include "toolchain/base/shared_value_stores.h"
#include "toolchain/base/value_store_impl.h"
#include "toolchain/base/yaml.h"
#include "toolchain/parse/node_ids.h"
#include "toolchain/sem_ir/constant.h"
#include "toolchain/sem_ir/dominance.h"
#include "toolchain/sem_ir/generic.h"
#include "toolchain/sem_ir/ids.h"
#include "toolchain/sem_ir/inst.h"
#include "toolchain/sem_ir/inst_kind.h"
@@ -158,8 +158,8 @@ auto File::Verify() const -> ErrorOr<Success> {
}
}
// TODO: Check that an instruction only references other instructions that are
// either global or that dominate it.
CARBON_RETURN_IF_ERROR(VerifyDominance(*this));
return Success();
}