mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-10-02 15:25:56 +01:00
Only change is to update the path to the fuzzer build extension. Original main commit message: > Add an initial parser library. (#30) > > This library builds a parse tree, very similar to a concrete syntax > tree. There are no semantics here, simply introducing the basic > syntactic structure. > > The current focus has been on the APIs and the data structures used to > represent the parse tree, and not on the actual code doing the > parsing. The code doing the parsing tries to be reasonably efficient > and reasonably easy to understand recursive descent parser. But there > is likely much that can be done to improve this code path. A notable > area where very little thought has been given yet are emitting good > diagnostics and doing good recovery in the event of parse errors. > > Also, this code does not try to match the current under-discussion > grammar closely. It is only partial and reflects discussions from some > time ago. It should be updated incrementally to reflect the current > expected grammar. > > The data structure used for the parse tree is unusual. The first > constraint is that there is a precise one-to-one correspondence > between the tokens produced by the lexer and the nodes in the parse > tree. Every token results in exactly one node. In that way, the parse > tree can be thought of as merely shaping the token stream into a tree. > > Each node is also represented with a fixed set of data that is densely > packed. Combined with the exact relationship to tokens, this allows us > to fully allocate the parse tree's storage, and to use a dense array > rather than a pointer-based tree structure. > > The tree structure itself is implicitly defined by tracking the size > of each subtree rooted at a particular node. See the code comments for > more details (and I'm happy to add more comments where necessary). The > goal is to minimize both the allocations (one), the working set size > of the tree as a whole, and optimize common iteration patterns. The > tree is stored in postorder. This allows depth-first postorder > iteration as well as topological iteration by walking in reverse. > > Building the parse tree in postorder is a natural consequence of the > grammar being LR rather than LL, which is a consequence of supporting > infix operators. > > As with the Lexer, the parser supports an API for operating on the > parse tree, as well as the ability to print the tree in both > a human-readable and machine-readable format (YAML-based). It includes > significant unit tests and a fuzz tester. The fuzzer's corpus will be > in a follow-up commit. > > This is the largest chunk of code already written by several of us > prior to open sourcing. (There are a few more pieces, but they are > significantly smaller and less interesting.) If there are major things > that folks would like to see happen here, it may make sense to move > them into issues for tracking. I have tried to update the code to > follow the style guidelines, but apologies if I missed anything, just > let me know. We also have issues #19 and #29 to track things that > already came up with the lexer. Co-authored-by: Jon Meow <46229924+jonmeow@users.noreply.github.com>
263 lines
9.9 KiB
C++
263 lines
9.9 KiB
C++
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
|
|
// Exceptions. See /LICENSE for license information.
|
|
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
|
|
|
#ifndef PARSER_PARSE_TEST_HELPERS_H_
|
|
#define PARSER_PARSE_TEST_HELPERS_H_
|
|
|
|
#include <ostream>
|
|
#include <string>
|
|
#include <vector>
|
|
|
|
#include "gmock/gmock.h"
|
|
#include "lexer/tokenized_buffer.h"
|
|
#include "llvm/ADT/STLExtras.h"
|
|
#include "llvm/ADT/SmallVector.h"
|
|
#include "llvm/ADT/StringRef.h"
|
|
#include "parser/parse_node_kind.h"
|
|
#include "parser/parse_tree.h"
|
|
|
|
namespace Carbon {
|
|
|
|
// Enable printing a parse tree from Google Mock.
|
|
inline void PrintTo(const ParseTree& tree, std::ostream* output) {
|
|
std::string text;
|
|
llvm::raw_string_ostream text_stream(text);
|
|
tree.Print(text_stream);
|
|
*output << "\n" << text_stream.str() << "\n";
|
|
}
|
|
|
|
namespace Testing {
|
|
|
|
// An aggregate used to describe an expected parse tree.
|
|
//
|
|
// This type is designed to be used via aggregate initialization with designated
|
|
// initializers. The latter make it easy to default everything and then override
|
|
// the desired aspects when writing an expectation in a test.
|
|
struct ExpectedNode {
|
|
ParseNodeKind kind = ParseNodeKind::EmptyDeclaration();
|
|
std::string text;
|
|
bool has_error = false;
|
|
bool skip_subtree = false;
|
|
std::vector<ExpectedNode> children;
|
|
};
|
|
|
|
// Implementation of a matcher for a parse tree based on a tree of expected
|
|
// nodes.
|
|
//
|
|
// Don't create this directly, instead use `MatchParseTreeNodes` to construct a
|
|
// matcher based on this.
|
|
class ExpectedNodesMatcher
|
|
: public ::testing::MatcherInterface<const ParseTree&> {
|
|
public:
|
|
ExpectedNodesMatcher(llvm::SmallVector<ExpectedNode, 0> expected_nodess)
|
|
: expected_nodes(std::move(expected_nodess)) {}
|
|
|
|
auto MatchAndExplain(const ParseTree& tree,
|
|
::testing::MatchResultListener* output_ptr) const
|
|
-> bool override;
|
|
auto DescribeTo(std::ostream* output_ptr) const -> void override;
|
|
|
|
private:
|
|
auto MatchExpectedNode(const ParseTree& tree, ParseTree::Node n,
|
|
int postorder_index, ExpectedNode expected_node,
|
|
::testing::MatchResultListener& output) const -> bool;
|
|
|
|
llvm::SmallVector<ExpectedNode, 0> expected_nodes;
|
|
};
|
|
|
|
// Implementation of the Google Mock interface for matching (and explaining any
|
|
// failure).
|
|
inline auto ExpectedNodesMatcher::MatchAndExplain(
|
|
const ParseTree& tree, ::testing::MatchResultListener* output_ptr) const
|
|
-> bool {
|
|
auto& output = *output_ptr;
|
|
bool matches = true;
|
|
const auto rpo = llvm::reverse(tree.Postorder());
|
|
const auto nodes_begin = rpo.begin();
|
|
const auto nodes_end = rpo.end();
|
|
auto nodes_it = nodes_begin;
|
|
llvm::SmallVector<const ExpectedNode*, 16> expected_node_stack;
|
|
for (const ExpectedNode& en : expected_nodes)
|
|
expected_node_stack.push_back(&en);
|
|
while (!expected_node_stack.empty()) {
|
|
if (nodes_it == nodes_end)
|
|
// We'll check the size outside the loop.
|
|
break;
|
|
|
|
ParseTree::Node n = *nodes_it++;
|
|
int postorder_index = n.GetIndex();
|
|
|
|
const ExpectedNode& expected_node = *expected_node_stack.pop_back_val();
|
|
|
|
if (!MatchExpectedNode(tree, n, postorder_index, expected_node, output))
|
|
matches = false;
|
|
|
|
if (expected_node.skip_subtree) {
|
|
assert(expected_node.children.empty() &&
|
|
"Must not skip an expected subtree while specifying expected "
|
|
"children!");
|
|
nodes_it = llvm::reverse(tree.Postorder(n)).end();
|
|
continue;
|
|
}
|
|
|
|
// We want to make sure we don't end up with unsynchronized walks, so skip
|
|
// ahead in the tree to ensure that the number of children of this node and
|
|
// the expected number of children match.
|
|
int num_children =
|
|
std::distance(tree.Children(n).begin(), tree.Children(n).end());
|
|
if (num_children != static_cast<int>(expected_node.children.size())) {
|
|
output
|
|
<< "\nParse node (postorder index #" << postorder_index << ") has "
|
|
<< num_children << " children, expected "
|
|
<< expected_node.children.size()
|
|
<< ". Skipping this subtree to avoid any unsynchronized tree walk.";
|
|
matches = false;
|
|
nodes_it = llvm::reverse(tree.Postorder(n)).end();
|
|
continue;
|
|
}
|
|
|
|
// Push the children onto the stack to continue matching. The expectation
|
|
// is in preorder, but we visit the parse tree in reverse postorder. This
|
|
// causes the siblings to be visited in reverse order from the expected
|
|
// list. However, we use a stack which inherently does this reverse for us
|
|
// so we simply append to the stack here.
|
|
for (const ExpectedNode& child_expected_node : expected_node.children)
|
|
expected_node_stack.push_back(&child_expected_node);
|
|
}
|
|
|
|
// We don't directly check the size because we allow expectations to skip
|
|
// subtrees. Instead, we need to check that we successfully processed all of
|
|
// the actual tree and consumed all of the expected tree.
|
|
if (nodes_it != nodes_end) {
|
|
assert(expected_node_stack.empty() &&
|
|
"If we have unmatched nodes in the input tree, should only finish "
|
|
"having fully processed expected tree.");
|
|
output << "\nFinished processing expected nodes and there are still "
|
|
<< (nodes_end - nodes_it) << " unexpected nodes.";
|
|
matches = false;
|
|
} else if (!expected_node_stack.empty()) {
|
|
output << "\nProcessed all " << (nodes_end - nodes_begin)
|
|
<< " nodes and still have " << expected_node_stack.size()
|
|
<< " expected nodes that were unmatched.";
|
|
matches = false;
|
|
}
|
|
|
|
return matches;
|
|
}
|
|
|
|
// Implementation of the Google Mock interface for describing the expected node
|
|
// tree.
|
|
//
|
|
// This is designed to describe the expected tree node structure in as similar
|
|
// of a format to the parse tree's print format as is reasonable. There is both
|
|
// more and less information, so it won't be exact, but should be close enough
|
|
// to make it easy to visually compare the two.
|
|
inline auto ExpectedNodesMatcher::DescribeTo(std::ostream* output_ptr) const
|
|
-> void {
|
|
auto& output = *output_ptr;
|
|
output << "Matches expected node pattern:\n[\n";
|
|
|
|
// We want to walk these in RPO instead of in preorder to match the printing
|
|
// of the actual parse tree.
|
|
llvm::SmallVector<std::pair<const ExpectedNode*, int>, 16>
|
|
expected_node_stack;
|
|
for (const ExpectedNode& expected_node : llvm::reverse(expected_nodes))
|
|
expected_node_stack.push_back({&expected_node, 0});
|
|
|
|
while (!expected_node_stack.empty()) {
|
|
const ExpectedNode& expected_node = *expected_node_stack.back().first;
|
|
int depth = expected_node_stack.back().second;
|
|
expected_node_stack.pop_back();
|
|
for (int indent_count = 0; indent_count < depth; ++indent_count)
|
|
output << " ";
|
|
output << "{kind: '" << expected_node.kind.GetName().str() << "'";
|
|
if (!expected_node.text.empty())
|
|
output << ", text: '" << expected_node.text << "'";
|
|
if (expected_node.has_error)
|
|
output << ", has_error: yes";
|
|
if (expected_node.skip_subtree)
|
|
output << ", skip_subtree: yes";
|
|
|
|
if (!expected_node.children.empty()) {
|
|
assert(!expected_node.skip_subtree &&
|
|
"Must not have children and skip a subtree!");
|
|
output << ", children: [\n";
|
|
for (const ExpectedNode& child_expected_node :
|
|
llvm::reverse(expected_node.children))
|
|
expected_node_stack.push_back({&child_expected_node, depth + 1});
|
|
// If we have children, we know we're not popping off.
|
|
continue;
|
|
}
|
|
|
|
// If this is some form of leaf we'll at least need to close it. It may also
|
|
// be the last sibling of its parent, and we'll need to close any parents as
|
|
// we pop up.
|
|
output << "}";
|
|
if (!expected_node_stack.empty()) {
|
|
assert(depth >= expected_node_stack.back().second &&
|
|
"Cannot have an increase in depth on a leaf node!");
|
|
// The distance we need to pop is the difference in depth.
|
|
int pop_depth = depth - expected_node_stack.back().second;
|
|
for (int pop_count = 0; pop_count < pop_depth; ++pop_count)
|
|
// Close both the children array and the node mapping.
|
|
output << "]}";
|
|
}
|
|
output << "\n";
|
|
}
|
|
output << "]\n";
|
|
}
|
|
|
|
inline auto ExpectedNodesMatcher::MatchExpectedNode(
|
|
const ParseTree& tree, ParseTree::Node n, int postorder_index,
|
|
ExpectedNode expected_node, ::testing::MatchResultListener& output) const
|
|
-> bool {
|
|
bool matches = true;
|
|
|
|
ParseNodeKind kind = tree.GetNodeKind(n);
|
|
if (kind != expected_node.kind) {
|
|
output << "\nParse node (postorder index #" << postorder_index << ") is a "
|
|
<< kind.GetName().str() << ", expected a "
|
|
<< expected_node.kind.GetName().str() << ".";
|
|
matches = false;
|
|
}
|
|
|
|
if (tree.HasErrorInNode(n) != expected_node.has_error) {
|
|
output << "\nParse node (postorder index #" << postorder_index << ") "
|
|
<< (tree.HasErrorInNode(n) ? "has an error"
|
|
: "does not have an error")
|
|
<< ", expected that it "
|
|
<< (expected_node.has_error ? "has an error"
|
|
: "does not have an error")
|
|
<< ".";
|
|
matches = false;
|
|
}
|
|
|
|
llvm::StringRef node_text = tree.GetNodeText(n);
|
|
if (!expected_node.text.empty() && node_text != expected_node.text) {
|
|
output << "\nParse node (postorder index #" << postorder_index
|
|
<< ") is spelled '" << node_text.str() << "', expected '"
|
|
<< expected_node.text << "'.";
|
|
matches = false;
|
|
}
|
|
|
|
return matches;
|
|
}
|
|
|
|
// Creates a matcher for a parse tree using a tree of expected nodes.
|
|
//
|
|
// This is intended to be used with an braced initializer list style aggregate
|
|
// initializer for an argument, allowing it to describe a tree structure via
|
|
// nested `ExpectedNode` objects.
|
|
inline auto MatchParseTreeNodes(
|
|
llvm::SmallVector<ExpectedNode, 0> expected_nodes)
|
|
-> ::testing::Matcher<const ParseTree&> {
|
|
return ::testing::MakeMatcher(
|
|
new ExpectedNodesMatcher(std::move(expected_nodes)));
|
|
}
|
|
|
|
} // namespace Testing
|
|
} // namespace Carbon
|
|
|
|
#endif // PARSER_PARSE_TEST_HELPERS_H_
|