mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-10-05 15:41:04 +01:00
Separate subtree size information from parse nodes. (#4174)
Move subtree sizes over to TreeAndSubtrees, using the different structure to represent the additional parse work that occurs, as well as making it clear which functions require the extra information. My intent is to make it hard to use this by accident. The subtree size is still tracked during Parse::Tree construction. I think a lot of that can be cleaned up, although we use it during placeholder assignment so it may take some work. I wanted to see what people thought about this before taking action on such a change. I'm using a 1m line source file generated by #4124 for testing. Command is `time bazel-bin/toolchain/install/prefix_root/bin/carbon compile --phase=check --dump-mem-usage ~/tmp/data.carbon` At head, what I'm seeing is: ``` ... parse_tree_.node_impls_: used_bytes: 61516116 reserved_bytes: 61516116 ... Total: used_bytes: 447814230 reserved_bytes: 551663894 ... 1.43s user 0.14s system 99% cpu 1.565 total ``` With `Tree::Verify` disabled completely, it looks like: ``` parse_tree_.node_impls_: used_bytes: 41010744 reserved_bytes: 41010744 ... Total: used_bytes: 427308858 reserved_bytes: 531158522 ... 1.20s user 0.13s system 99% cpu 1.332 total ``` Re-enabling just the basic verification (what is now `Tree::Verify`), I'm seeing maybe 0.05s slower, but that's within noise for my system. I do see variability in my timing results, and overall I think this is a 0.2s +/- 0.1s improvement versus the earlier (always testing `Extract` code) implementation. That's opt; debug builds will be unaffected, because the same checking occurs as before. Note, the subtree size is a third of the node representation, which is why I'm showing the decrease in memory usage here.
This commit is contained in:
+22
-245
@@ -10,6 +10,7 @@
|
||||
#include "llvm/ADT/SmallVector.h"
|
||||
#include "toolchain/lex/tokenized_buffer.h"
|
||||
#include "toolchain/parse/node_kind.h"
|
||||
#include "toolchain/parse/tree_and_subtrees.h"
|
||||
#include "toolchain/parse/typed_nodes.h"
|
||||
|
||||
namespace Carbon::Parse {
|
||||
@@ -20,28 +21,6 @@ auto Tree::postorder() const -> llvm::iterator_range<PostorderIterator> {
|
||||
PostorderIterator(NodeId(node_impls_.size())));
|
||||
}
|
||||
|
||||
auto Tree::postorder(NodeId n) const
|
||||
-> llvm::iterator_range<PostorderIterator> {
|
||||
// The postorder ends after this node, the root, and begins at the start of
|
||||
// its subtree.
|
||||
int start_index = n.index - node_impls_[n.index].subtree_size + 1;
|
||||
return PostorderIterator::MakeRange(NodeId(start_index), n);
|
||||
}
|
||||
|
||||
auto Tree::children(NodeId n) const -> llvm::iterator_range<SiblingIterator> {
|
||||
CARBON_CHECK(n.is_valid());
|
||||
int end_index = n.index - node_impls_[n.index].subtree_size;
|
||||
return llvm::iterator_range<SiblingIterator>(
|
||||
SiblingIterator(*this, NodeId(n.index - 1)),
|
||||
SiblingIterator(*this, NodeId(end_index)));
|
||||
}
|
||||
|
||||
auto Tree::roots() const -> llvm::iterator_range<SiblingIterator> {
|
||||
return llvm::iterator_range<SiblingIterator>(
|
||||
SiblingIterator(*this, NodeId(static_cast<int>(node_impls_.size()) - 1)),
|
||||
SiblingIterator(*this, NodeId(-1)));
|
||||
}
|
||||
|
||||
auto Tree::node_has_error(NodeId n) const -> bool {
|
||||
CARBON_CHECK(n.is_valid());
|
||||
return node_impls_[n.index].has_error;
|
||||
@@ -57,245 +36,47 @@ auto Tree::node_token(NodeId n) const -> Lex::TokenIndex {
|
||||
return node_impls_[n.index].token;
|
||||
}
|
||||
|
||||
auto Tree::node_subtree_size(NodeId n) const -> int32_t {
|
||||
CARBON_CHECK(n.is_valid());
|
||||
return node_impls_[n.index].subtree_size;
|
||||
}
|
||||
|
||||
auto Tree::PrintNode(llvm::raw_ostream& output, NodeId n, int depth,
|
||||
bool preorder) const -> bool {
|
||||
const auto& n_impl = node_impls_[n.index];
|
||||
output.indent(2 * (depth + 2));
|
||||
output << "{";
|
||||
// If children are being added, include node_index in order to disambiguate
|
||||
// nodes.
|
||||
if (preorder) {
|
||||
output << "node_index: " << n << ", ";
|
||||
}
|
||||
output << "kind: '" << n_impl.kind << "', text: '"
|
||||
<< tokens_->GetTokenText(n_impl.token) << "'";
|
||||
|
||||
if (n_impl.has_error) {
|
||||
output << ", has_error: yes";
|
||||
}
|
||||
|
||||
if (n_impl.subtree_size > 1) {
|
||||
output << ", subtree_size: " << n_impl.subtree_size;
|
||||
if (preorder) {
|
||||
output << ", children: [\n";
|
||||
return true;
|
||||
}
|
||||
}
|
||||
output << "}";
|
||||
return false;
|
||||
}
|
||||
|
||||
auto Tree::Print(llvm::raw_ostream& output) const -> void {
|
||||
output << "- filename: " << tokens_->source().filename() << "\n"
|
||||
<< " parse_tree: [\n";
|
||||
|
||||
// Walk the tree just to calculate depths for each node.
|
||||
llvm::SmallVector<int> indents;
|
||||
indents.append(size(), 0);
|
||||
|
||||
llvm::SmallVector<std::pair<NodeId, int>, 16> node_stack;
|
||||
for (NodeId n : roots()) {
|
||||
node_stack.push_back({n, 0});
|
||||
}
|
||||
|
||||
while (!node_stack.empty()) {
|
||||
NodeId n = NodeId::Invalid;
|
||||
int depth;
|
||||
std::tie(n, depth) = node_stack.pop_back_val();
|
||||
for (NodeId sibling_n : children(n)) {
|
||||
indents[sibling_n.index] = depth + 1;
|
||||
node_stack.push_back({sibling_n, depth + 1});
|
||||
}
|
||||
}
|
||||
|
||||
for (NodeId n : postorder()) {
|
||||
PrintNode(output, n, indents[n.index], /*preorder=*/false);
|
||||
output << ",\n";
|
||||
}
|
||||
output << " ]\n";
|
||||
}
|
||||
|
||||
auto Tree::Print(llvm::raw_ostream& output, bool preorder) const -> void {
|
||||
if (!preorder) {
|
||||
Print(output);
|
||||
return;
|
||||
}
|
||||
|
||||
output << "- filename: " << tokens_->source().filename() << "\n"
|
||||
<< " parse_tree: [\n";
|
||||
|
||||
// The parse tree is stored in postorder. The preorder can be constructed
|
||||
// by reversing the order of each level of siblings within an RPO. The
|
||||
// sibling iterators are directly built around RPO and so can be used with a
|
||||
// stack to produce preorder.
|
||||
|
||||
// The roots, like siblings, are in RPO (so reversed), but we add them in
|
||||
// order here because we'll pop off the stack effectively reversing then.
|
||||
llvm::SmallVector<std::pair<NodeId, int>, 16> node_stack;
|
||||
for (NodeId n : roots()) {
|
||||
node_stack.push_back({n, 0});
|
||||
}
|
||||
|
||||
while (!node_stack.empty()) {
|
||||
NodeId n = NodeId::Invalid;
|
||||
int depth;
|
||||
std::tie(n, depth) = node_stack.pop_back_val();
|
||||
|
||||
if (PrintNode(output, n, depth, /*preorder=*/true)) {
|
||||
// Has children, so we descend. We append the children in order here as
|
||||
// well because they will get reversed when popped off the stack.
|
||||
for (NodeId sibling_n : children(n)) {
|
||||
node_stack.push_back({sibling_n, depth + 1});
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
int next_depth = node_stack.empty() ? 0 : node_stack.back().second;
|
||||
CARBON_CHECK(next_depth <= depth) << "Cannot have the next depth increase!";
|
||||
for (int close_children_count : llvm::seq(0, depth - next_depth)) {
|
||||
(void)close_children_count;
|
||||
output << "]}";
|
||||
}
|
||||
|
||||
// We always end with a comma and a new line as we'll move to the next
|
||||
// node at whatever the current level ends up being.
|
||||
output << " ,\n";
|
||||
}
|
||||
output << " ]\n";
|
||||
}
|
||||
|
||||
auto Tree::CollectMemUsage(MemUsage& mem_usage, llvm::StringRef label) const
|
||||
-> void {
|
||||
mem_usage.Add(MemUsage::ConcatLabel(label, "node_impls_"), node_impls_);
|
||||
mem_usage.Add(MemUsage::ConcatLabel(label, "imports_"), imports_);
|
||||
}
|
||||
|
||||
auto Tree::VerifyExtract(NodeId node_id, NodeKind kind,
|
||||
ErrorBuilder* trace) const -> bool {
|
||||
switch (kind) {
|
||||
#define CARBON_PARSE_NODE_KIND(Name) \
|
||||
case NodeKind::Name: \
|
||||
return VerifyExtractAs<Name>(node_id, trace).has_value();
|
||||
#include "toolchain/parse/node_kind.def"
|
||||
}
|
||||
TreeAndSubtrees(*tokens_, *this).Print(output);
|
||||
}
|
||||
|
||||
auto Tree::Verify() const -> ErrorOr<Success> {
|
||||
llvm::SmallVector<NodeId> nodes;
|
||||
// Traverse the tree in postorder.
|
||||
for (NodeId n : postorder()) {
|
||||
const auto& n_impl = node_impls_[n.index];
|
||||
|
||||
if (n_impl.has_error && !has_errors_) {
|
||||
if (node_has_error(n) && !has_errors()) {
|
||||
return Error(llvm::formatv(
|
||||
"NodeId #{0} has errors, but the tree is not marked as having any.",
|
||||
n.index));
|
||||
"Node {0} has errors, but the tree is not marked as having any.", n));
|
||||
}
|
||||
|
||||
if (n_impl.kind == NodeKind::Placeholder) {
|
||||
if (node_kind(n) == NodeKind::Placeholder) {
|
||||
return Error(llvm::formatv(
|
||||
"Node #{0} is a placeholder node that wasn't replaced.", n.index));
|
||||
"Node {0} is a placeholder node that wasn't replaced.", n));
|
||||
}
|
||||
// Should extract successfully if node not marked as having an error.
|
||||
// Without this code, a 10 mloc test case of lex & parse takes
|
||||
// 4.129 s ± 0.041 s. With this additional verification, it takes
|
||||
// 5.768 s ± 0.036 s.
|
||||
if (!n_impl.has_error && !VerifyExtract(n, n_impl.kind, nullptr)) {
|
||||
ErrorBuilder trace;
|
||||
trace << llvm::formatv(
|
||||
"NodeId #{0} couldn't be extracted as a {1}. Trace:\n", n,
|
||||
n_impl.kind);
|
||||
VerifyExtract(n, n_impl.kind, &trace);
|
||||
return trace;
|
||||
}
|
||||
|
||||
int subtree_size = 1;
|
||||
if (n_impl.kind.has_bracket()) {
|
||||
int child_count = 0;
|
||||
while (true) {
|
||||
if (nodes.empty()) {
|
||||
return Error(
|
||||
llvm::formatv("NodeId #{0} is a {1} with bracket {2}, but didn't "
|
||||
"find the bracket.",
|
||||
n, n_impl.kind, n_impl.kind.bracket()));
|
||||
}
|
||||
auto child_impl = node_impls_[nodes.pop_back_val().index];
|
||||
subtree_size += child_impl.subtree_size;
|
||||
++child_count;
|
||||
if (n_impl.kind.bracket() == child_impl.kind) {
|
||||
// If there's a bracketing node and a child count, verify the child
|
||||
// count too.
|
||||
if (n_impl.kind.has_child_count() &&
|
||||
child_count != n_impl.kind.child_count()) {
|
||||
return Error(llvm::formatv(
|
||||
"NodeId #{0} is a {1} with child_count {2}, but encountered "
|
||||
"{3} nodes before we reached the bracketing node.",
|
||||
n, n_impl.kind, n_impl.kind.child_count(), child_count));
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (int i : llvm::seq(n_impl.kind.child_count())) {
|
||||
if (nodes.empty()) {
|
||||
return Error(llvm::formatv(
|
||||
"NodeId #{0} is a {1} with child_count {2}, but only had {3} "
|
||||
"nodes to consume.",
|
||||
n, n_impl.kind, n_impl.kind.child_count(), i));
|
||||
}
|
||||
auto child_impl = node_impls_[nodes.pop_back_val().index];
|
||||
subtree_size += child_impl.subtree_size;
|
||||
}
|
||||
}
|
||||
if (n_impl.subtree_size != subtree_size) {
|
||||
return Error(llvm::formatv(
|
||||
"NodeId #{0} is a {1} with subtree_size of {2}, but calculated {3}.",
|
||||
n, n_impl.kind, n_impl.subtree_size, subtree_size));
|
||||
}
|
||||
nodes.push_back(n);
|
||||
}
|
||||
|
||||
// Remaining nodes should all be roots in the tree; make sure they line up.
|
||||
CARBON_CHECK(nodes.back().index ==
|
||||
static_cast<int32_t>(node_impls_.size()) - 1)
|
||||
<< nodes.back() << " " << node_impls_.size() - 1;
|
||||
int prev_index = -1;
|
||||
for (const auto& n : nodes) {
|
||||
const auto& n_impl = node_impls_[n.index];
|
||||
|
||||
if (n.index - n_impl.subtree_size != prev_index) {
|
||||
return Error(
|
||||
llvm::formatv("NodeId #{0} is a root {1} with subtree_size {2}, but "
|
||||
"previous root was at #{3}.",
|
||||
n, n_impl.kind, n_impl.subtree_size, prev_index));
|
||||
}
|
||||
prev_index = n.index;
|
||||
if (!has_errors() &&
|
||||
static_cast<int32_t>(size()) != tokens_->expected_parse_tree_size()) {
|
||||
return Error(llvm::formatv(
|
||||
"Tree has {0} nodes and no errors, but "
|
||||
"Lex::TokenizedBuffer expected {1} nodes for {2} tokens.",
|
||||
size(), tokens_->expected_parse_tree_size(), tokens_->size()));
|
||||
}
|
||||
|
||||
// Validate the roots, ensures Tree::ExtractFile() doesn't CHECK-fail.
|
||||
if (!TryExtractNodeFromChildren<File>(NodeId::Invalid, roots(), nullptr)) {
|
||||
ErrorBuilder trace;
|
||||
trace << "Roots of tree couldn't be extracted as a `File`. Trace:\n";
|
||||
TryExtractNodeFromChildren<File>(NodeId::Invalid, roots(), &trace);
|
||||
return trace;
|
||||
}
|
||||
#ifndef NDEBUG
|
||||
TreeAndSubtrees subtrees(*tokens_, *this);
|
||||
CARBON_RETURN_IF_ERROR(subtrees.Verify());
|
||||
#endif // NDEBUG
|
||||
|
||||
if (!has_errors_ && static_cast<int32_t>(node_impls_.size()) !=
|
||||
tokens_->expected_parse_tree_size()) {
|
||||
return Error(
|
||||
llvm::formatv("Tree has {0} nodes and no errors, but "
|
||||
"Lex::TokenizedBuffer expected {1} nodes for {2} tokens.",
|
||||
node_impls_.size(), tokens_->expected_parse_tree_size(),
|
||||
tokens_->size()));
|
||||
}
|
||||
return Success();
|
||||
}
|
||||
|
||||
auto Tree::CollectMemUsage(MemUsage& mem_usage, llvm::StringRef label) const
|
||||
-> void {
|
||||
mem_usage.Add(MemUsage::ConcatLabel(label, "node_impls_"), node_impls_);
|
||||
mem_usage.Add(MemUsage::ConcatLabel(label, "imports_"), imports_);
|
||||
}
|
||||
|
||||
auto Tree::PostorderIterator::MakeRange(NodeId begin, NodeId end)
|
||||
-> llvm::iterator_range<PostorderIterator> {
|
||||
CARBON_CHECK(begin.is_valid() && end.is_valid());
|
||||
@@ -307,8 +88,4 @@ auto Tree::PostorderIterator::Print(llvm::raw_ostream& output) const -> void {
|
||||
output << node_;
|
||||
}
|
||||
|
||||
auto Tree::SiblingIterator::Print(llvm::raw_ostream& output) const -> void {
|
||||
output << node_;
|
||||
}
|
||||
|
||||
} // namespace Carbon::Parse
|
||||
|
||||
Reference in New Issue
Block a user