Separate subtree size information from parse nodes. (#4174)

Move subtree sizes over to TreeAndSubtrees, using the different
structure to represent the additional parse work that occurs, as well as
making it clear which functions require the extra information. My intent
is to make it hard to use this by accident.

The subtree size is still tracked during Parse::Tree construction. I
think a lot of that can be cleaned up, although we use it during
placeholder assignment so it may take some work. I wanted to see what
people thought about this before taking action on such a change.

I'm using a 1m line source file generated by #4124 for testing. Command
is `time bazel-bin/toolchain/install/prefix_root/bin/carbon compile
--phase=check --dump-mem-usage ~/tmp/data.carbon`

At head, what I'm seeing is:

```
...
parse_tree_.node_impls_:
  used_bytes:      61516116
  reserved_bytes:  61516116
...
Total:
  used_bytes:      447814230
  reserved_bytes:  551663894
...
1.43s user 0.14s system 99% cpu 1.565 total
```

With `Tree::Verify` disabled completely, it looks like:
```
parse_tree_.node_impls_:
  used_bytes:      41010744
  reserved_bytes:  41010744
...
Total:
  used_bytes:      427308858
  reserved_bytes:  531158522
...
1.20s user 0.13s system 99% cpu 1.332 total
```

Re-enabling just the basic verification (what is now `Tree::Verify`),
I'm seeing maybe 0.05s slower, but that's within noise for my system. I
do see variability in my timing results, and overall I think this is a
0.2s +/- 0.1s improvement versus the earlier (always testing `Extract`
code) implementation. That's opt; debug builds will be unaffected,
because the same checking occurs as before.

Note, the subtree size is a third of the node representation, which is
why I'm showing the decrease in memory usage here.
This commit is contained in:
Jon Ross-Perkins
2024-07-31 19:39:45 +00:00
committed by GitHub
parent e6e61e14ae
commit f67791cfee
40 changed files with 767 additions and 692 deletions
+24 -19
View File
@@ -9,6 +9,7 @@
#include "common/error.h"
#include "common/struct_reflection.h"
#include "toolchain/parse/tree.h"
#include "toolchain/parse/tree_and_subtrees.h"
#include "toolchain/parse/typed_nodes.h"
namespace Carbon::Parse {
@@ -20,12 +21,12 @@ namespace {
class NodeExtractor {
public:
struct CheckpointState {
Tree::SiblingIterator it;
TreeAndSubtrees::SiblingIterator it;
};
NodeExtractor(const Tree* tree, Lex::TokenizedBuffer* tokens,
NodeExtractor(const TreeAndSubtrees* tree, const Lex::TokenizedBuffer* tokens,
ErrorBuilder* trace, NodeId node_id,
llvm::iterator_range<Tree::SiblingIterator> children)
llvm::iterator_range<TreeAndSubtrees::SiblingIterator> children)
: tree_(tree),
tokens_(tokens),
trace_(trace),
@@ -34,9 +35,11 @@ class NodeExtractor {
end_(children.end()) {}
auto at_end() const -> bool { return it_ == end_; }
auto kind() const -> NodeKind { return tree_->node_kind(*it_); }
auto kind() const -> NodeKind { return tree_->tree().node_kind(*it_); }
auto has_token() const -> bool { return node_id_.is_valid(); }
auto token() const -> Lex::TokenIndex { return tree_->node_token(node_id_); }
auto token() const -> Lex::TokenIndex {
return tree_->tree().node_token(node_id_);
}
auto token_kind() const -> Lex::TokenKind {
return tokens_->GetKind(token());
}
@@ -73,12 +76,12 @@ class NodeExtractor {
std::tuple<U...>* /*type*/) -> std::optional<T>;
private:
const Tree* tree_;
Lex::TokenizedBuffer* tokens_;
const TreeAndSubtrees* tree_;
const Lex::TokenizedBuffer* tokens_;
ErrorBuilder* trace_;
NodeId node_id_;
Tree::SiblingIterator it_;
Tree::SiblingIterator end_;
TreeAndSubtrees::SiblingIterator it_;
TreeAndSubtrees::SiblingIterator end_;
};
} // namespace
@@ -97,8 +100,8 @@ namespace {
// };
// ```
//
// Note that `Tree::SiblingIterator`s iterate in reverse order through the
// children of a node.
// Note that `TreeAndSubtrees::SiblingIterator`s iterate in reverse order
// through the children of a node.
//
// This class is only in this file.
template <typename T>
@@ -320,7 +323,7 @@ auto NodeExtractor::MatchesTokenKind(Lex::TokenKind expected_kind) const
if (token_kind() != expected_kind) {
if (trace_) {
*trace_ << "Token " << expected_kind << " expected for "
<< tree_->node_kind(node_id_) << ", found " << token_kind()
<< tree_->tree().node_kind(node_id_) << ", found " << token_kind()
<< "\n";
}
return false;
@@ -405,14 +408,15 @@ struct Extractable {
} // namespace
template <typename T>
auto Tree::TryExtractNodeFromChildren(
NodeId node_id, llvm::iterator_range<Tree::SiblingIterator> children,
auto TreeAndSubtrees::TryExtractNodeFromChildren(
NodeId node_id,
llvm::iterator_range<TreeAndSubtrees::SiblingIterator> children,
ErrorBuilder* trace) const -> std::optional<T> {
NodeExtractor extractor(this, tokens_, trace, node_id, children);
auto result = Extractable<T>::ExtractImpl(extractor);
if (!extractor.at_end()) {
if (trace) {
*trace << "Error: " << node_kind(extractor.ExtractNode())
*trace << "Error: " << tree_->node_kind(extractor.ExtractNode())
<< " node left unconsumed.";
}
return std::nullopt;
@@ -421,16 +425,17 @@ auto Tree::TryExtractNodeFromChildren(
}
// Manually instantiate Tree::TryExtractNodeFromChildren
#define CARBON_PARSE_NODE_KIND(KindName) \
template auto Tree::TryExtractNodeFromChildren<KindName>( \
NodeId node_id, llvm::iterator_range<Tree::SiblingIterator> children, \
#define CARBON_PARSE_NODE_KIND(KindName) \
template auto TreeAndSubtrees::TryExtractNodeFromChildren<KindName>( \
NodeId node_id, \
llvm::iterator_range<TreeAndSubtrees::SiblingIterator> children, \
ErrorBuilder * trace) const -> std::optional<KindName>;
// Also instantiate for `File`, even though it isn't a parse node.
CARBON_PARSE_NODE_KIND(File)
#include "toolchain/parse/node_kind.def"
auto Tree::ExtractFile() const -> File {
auto TreeAndSubtrees::ExtractFile() const -> File {
return ExtractNodeFromChildren<File>(NodeId::Invalid, roots());
}