mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-10-05 22:02:55 +01:00
Initial support for expression parsing. (#455)
This commit is contained in:
@@ -13,14 +13,29 @@
|
||||
#error "Must define the x-macro to use this file."
|
||||
#endif
|
||||
|
||||
CARBON_PARSE_NODE_KIND(CodeBlockEnd)
|
||||
CARBON_PARSE_NODE_KIND(CodeBlock)
|
||||
// Declarations.
|
||||
CARBON_PARSE_NODE_KIND(DeclarationEnd)
|
||||
CARBON_PARSE_NODE_KIND(EmptyDeclaration)
|
||||
CARBON_PARSE_NODE_KIND(FunctionDeclaration)
|
||||
CARBON_PARSE_NODE_KIND(Identifier)
|
||||
CARBON_PARSE_NODE_KIND(DeclaredName)
|
||||
CARBON_PARSE_NODE_KIND(ParameterListEnd)
|
||||
CARBON_PARSE_NODE_KIND(ParameterList)
|
||||
CARBON_PARSE_NODE_KIND(FileEnd)
|
||||
|
||||
// Statements.
|
||||
CARBON_PARSE_NODE_KIND(CodeBlockEnd)
|
||||
CARBON_PARSE_NODE_KIND(CodeBlock)
|
||||
CARBON_PARSE_NODE_KIND(ExpressionStatement)
|
||||
|
||||
// Expressions.
|
||||
CARBON_PARSE_NODE_KIND(Literal)
|
||||
CARBON_PARSE_NODE_KIND(NameReference)
|
||||
CARBON_PARSE_NODE_KIND(ParenExpression)
|
||||
CARBON_PARSE_NODE_KIND(ParenExpressionEnd)
|
||||
CARBON_PARSE_NODE_KIND(DesignatorExpression)
|
||||
CARBON_PARSE_NODE_KIND(DesignatedName)
|
||||
CARBON_PARSE_NODE_KIND(CallExpression)
|
||||
CARBON_PARSE_NODE_KIND(CallExpressionComma)
|
||||
CARBON_PARSE_NODE_KIND(CallExpressionEnd)
|
||||
|
||||
#undef CARBON_PARSE_NODE_KIND
|
||||
|
||||
+84
-18
@@ -20,6 +20,7 @@
|
||||
namespace Carbon {
|
||||
namespace {
|
||||
|
||||
using Carbon::Testing::ExpectedNode;
|
||||
using Carbon::Testing::IsKeyValueScalars;
|
||||
using Carbon::Testing::MatchParseTreeNodes;
|
||||
using namespace Carbon::Testing::NodeMatchers;
|
||||
@@ -91,7 +92,7 @@ TEST_F(ParseTreeTest, BasicFunctionDeclaration) {
|
||||
EXPECT_THAT(tree,
|
||||
MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(
|
||||
"fn", MatchIdentifier("F"),
|
||||
"fn", MatchDeclaredName("F"),
|
||||
MatchParameterList("(", MatchParameterListEnd(")")),
|
||||
MatchDeclarationEnd(";")),
|
||||
MatchFileEnd()}));
|
||||
@@ -134,9 +135,10 @@ TEST_F(ParseTreeTest, FunctionDeclarationWithNoSignatureOrSemi) {
|
||||
TokenizedBuffer tokens = GetTokenizedBuffer("fn foo");
|
||||
ParseTree tree = ParseTree::Parse(tokens, consumer);
|
||||
EXPECT_TRUE(tree.HasErrors());
|
||||
EXPECT_THAT(tree, MatchParseTreeNodes({MatchFunctionDeclaration(
|
||||
HasError, MatchIdentifier("foo")),
|
||||
MatchFileEnd()}));
|
||||
EXPECT_THAT(tree,
|
||||
MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(HasError, MatchDeclaredName("foo")),
|
||||
MatchFileEnd()}));
|
||||
}
|
||||
|
||||
TEST_F(ParseTreeTest,
|
||||
@@ -145,7 +147,7 @@ TEST_F(ParseTreeTest,
|
||||
ParseTree tree = ParseTree::Parse(tokens, consumer);
|
||||
EXPECT_TRUE(tree.HasErrors());
|
||||
EXPECT_THAT(tree, MatchParseTreeNodes({MatchFunctionDeclaration(
|
||||
HasError, MatchIdentifier("foo"),
|
||||
HasError, MatchDeclaredName("foo"),
|
||||
MatchDeclarationEnd()),
|
||||
MatchFileEnd()}));
|
||||
}
|
||||
@@ -159,7 +161,7 @@ TEST_F(ParseTreeTest, FunctionDeclarationWithSingleIdentifierParameterList) {
|
||||
EXPECT_THAT(tree,
|
||||
MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(
|
||||
HasError, MatchIdentifier("foo"),
|
||||
HasError, MatchDeclaredName("foo"),
|
||||
MatchParameterList(HasError, MatchParameterListEnd()),
|
||||
MatchDeclarationEnd()),
|
||||
MatchFileEnd()}));
|
||||
@@ -195,7 +197,7 @@ TEST_F(ParseTreeTest, FunctionDeclarationSkipToNewlineWithoutSemi) {
|
||||
tree,
|
||||
MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(HasError),
|
||||
MatchFunctionDeclaration(MatchIdentifier("F"),
|
||||
MatchFunctionDeclaration(MatchDeclaredName("F"),
|
||||
MatchParameterList(MatchParameterListEnd()),
|
||||
MatchDeclarationEnd()),
|
||||
MatchFileEnd()}));
|
||||
@@ -213,7 +215,7 @@ TEST_F(ParseTreeTest, FunctionDeclarationSkipIndentedNewlineWithSemi) {
|
||||
tree,
|
||||
MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(HasError, MatchDeclarationEnd()),
|
||||
MatchFunctionDeclaration(MatchIdentifier("F"),
|
||||
MatchFunctionDeclaration(MatchDeclaredName("F"),
|
||||
MatchParameterList(MatchParameterListEnd()),
|
||||
MatchDeclarationEnd()),
|
||||
MatchFileEnd()}));
|
||||
@@ -231,7 +233,7 @@ TEST_F(ParseTreeTest, FunctionDeclarationSkipIndentedNewlineWithoutSemi) {
|
||||
tree,
|
||||
MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(HasError),
|
||||
MatchFunctionDeclaration(MatchIdentifier("F"),
|
||||
MatchFunctionDeclaration(MatchDeclaredName("F"),
|
||||
MatchParameterList(MatchParameterListEnd()),
|
||||
MatchDeclarationEnd()),
|
||||
MatchFileEnd()}));
|
||||
@@ -249,7 +251,7 @@ TEST_F(ParseTreeTest, FunctionDeclarationSkipIndentedNewlineUntilOutdent) {
|
||||
tree,
|
||||
MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(HasError),
|
||||
MatchFunctionDeclaration(MatchIdentifier("F"),
|
||||
MatchFunctionDeclaration(MatchDeclaredName("F"),
|
||||
MatchParameterList(MatchParameterListEnd()),
|
||||
MatchDeclarationEnd()),
|
||||
MatchFileEnd()}));
|
||||
@@ -276,7 +278,7 @@ TEST_F(ParseTreeTest, BasicFunctionDefinition) {
|
||||
EXPECT_FALSE(tree.HasErrors());
|
||||
EXPECT_THAT(tree, MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(
|
||||
MatchIdentifier("F"),
|
||||
MatchDeclaredName("F"),
|
||||
MatchParameterList(MatchParameterListEnd()),
|
||||
MatchCodeBlock("{", MatchCodeBlockEnd("}"))),
|
||||
MatchFileEnd()}));
|
||||
@@ -294,7 +296,7 @@ TEST_F(ParseTreeTest, FunctionDefinitionWithNestedBlocks) {
|
||||
EXPECT_THAT(
|
||||
tree, MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(
|
||||
MatchIdentifier("F"),
|
||||
MatchDeclaredName("F"),
|
||||
MatchParameterList(MatchParameterListEnd()),
|
||||
MatchCodeBlock(
|
||||
MatchCodeBlock(
|
||||
@@ -316,9 +318,10 @@ TEST_F(ParseTreeTest, FunctionDefinitionWithIdenifierInStatements) {
|
||||
EXPECT_TRUE(tree.HasErrors());
|
||||
EXPECT_THAT(tree, MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(
|
||||
MatchIdentifier("F"),
|
||||
MatchDeclaredName("F"),
|
||||
MatchParameterList(MatchParameterListEnd()),
|
||||
MatchCodeBlock(HasError, MatchCodeBlockEnd())),
|
||||
MatchCodeBlock(HasError, MatchNameReference("bar"),
|
||||
MatchCodeBlockEnd())),
|
||||
MatchFileEnd()}));
|
||||
}
|
||||
|
||||
@@ -331,13 +334,76 @@ TEST_F(ParseTreeTest, FunctionDefinitionWithIdenifierInNestedBlock) {
|
||||
// Note: this might become valid depending on the expression syntax. This test
|
||||
// shouldn't be taken as a sign it should remain invalid.
|
||||
EXPECT_TRUE(tree.HasErrors());
|
||||
EXPECT_THAT(tree,
|
||||
MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(
|
||||
MatchDeclaredName("F"),
|
||||
MatchParameterList(MatchParameterListEnd()),
|
||||
MatchCodeBlock(
|
||||
MatchCodeBlock(HasError, MatchNameReference("bar"),
|
||||
MatchCodeBlockEnd()),
|
||||
MatchCodeBlockEnd())),
|
||||
MatchFileEnd()}));
|
||||
}
|
||||
|
||||
TEST_F(ParseTreeTest, FunctionDefinitionWithFunctionCall) {
|
||||
TokenizedBuffer tokens = GetTokenizedBuffer(
|
||||
"fn F() {\n"
|
||||
" a.b.f(c.d, (e)).g();\n"
|
||||
"}");
|
||||
ParseTree tree = ParseTree::Parse(tokens, consumer);
|
||||
EXPECT_FALSE(tree.HasErrors());
|
||||
|
||||
ExpectedNode call_to_f = MatchCallExpression(
|
||||
MatchDesignatorExpression(
|
||||
MatchDesignatorExpression(MatchNameReference("a"),
|
||||
MatchDesignatedName("b")),
|
||||
MatchDesignatedName("f")),
|
||||
MatchDesignatorExpression(MatchNameReference("c"),
|
||||
MatchDesignatedName("d")),
|
||||
MatchCallExpressionComma(),
|
||||
MatchParenExpression(MatchNameReference("e"), MatchParenExpressionEnd()),
|
||||
MatchCallExpressionEnd());
|
||||
ExpectedNode statement = MatchExpressionStatement(MatchCallExpression(
|
||||
MatchDesignatorExpression(call_to_f, MatchDesignatedName("g")),
|
||||
MatchCallExpressionEnd()));
|
||||
|
||||
EXPECT_THAT(tree, MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(
|
||||
MatchDeclaredName("F"),
|
||||
MatchParameterList(MatchParameterListEnd()),
|
||||
MatchCodeBlock(statement, MatchCodeBlockEnd())),
|
||||
MatchFileEnd()}));
|
||||
}
|
||||
|
||||
TEST_F(ParseTreeTest, InvalidDesignators) {
|
||||
TokenizedBuffer tokens = GetTokenizedBuffer(
|
||||
"fn F() {\n"
|
||||
" a.;\n"
|
||||
" a.fn;\n"
|
||||
" a.42;\n"
|
||||
"}");
|
||||
ParseTree tree = ParseTree::Parse(tokens, consumer);
|
||||
EXPECT_TRUE(tree.HasErrors());
|
||||
|
||||
EXPECT_THAT(
|
||||
tree,
|
||||
MatchParseTreeNodes(
|
||||
{MatchFunctionDeclaration(
|
||||
MatchIdentifier("F"),
|
||||
MatchDeclaredName("F"),
|
||||
MatchParameterList(MatchParameterListEnd()),
|
||||
MatchCodeBlock(MatchCodeBlock(HasError, MatchCodeBlockEnd()),
|
||||
MatchCodeBlock(MatchExpressionStatement(
|
||||
MatchDesignatorExpression(
|
||||
MatchNameReference("a"), ".", HasError),
|
||||
";"),
|
||||
MatchExpressionStatement(
|
||||
MatchDesignatorExpression(
|
||||
MatchNameReference("a"), ".", HasError),
|
||||
";"),
|
||||
MatchExpressionStatement(
|
||||
MatchDesignatorExpression(
|
||||
MatchNameReference("a"), ".", HasError),
|
||||
HasError, ";"),
|
||||
MatchCodeBlockEnd())),
|
||||
MatchFileEnd()}));
|
||||
}
|
||||
@@ -368,7 +434,7 @@ TEST_F(ParseTreeTest, Printing) {
|
||||
StrEq("{node_index: 4, kind: 'FunctionDeclaration', text: 'fn', "
|
||||
"subtree_size: 5, children: ["));
|
||||
EXPECT_THAT(GetAndDropLine(print),
|
||||
StrEq(" {node_index: 0, kind: 'Identifier', text: 'F'},"));
|
||||
StrEq(" {node_index: 0, kind: 'DeclaredName', text: 'F'},"));
|
||||
EXPECT_THAT(GetAndDropLine(print),
|
||||
StrEq(" {node_index: 2, kind: 'ParameterList', text: '(', "
|
||||
"subtree_size: 2, children: ["));
|
||||
@@ -432,7 +498,7 @@ TEST_F(ParseTreeTest, PrintingAsYAML) {
|
||||
auto ckvi = node->begin();
|
||||
EXPECT_THAT(&*ckvi, IsKeyValueScalars("node_index", "0"));
|
||||
++ckvi;
|
||||
EXPECT_THAT(&*ckvi, IsKeyValueScalars("kind", "Identifier"));
|
||||
EXPECT_THAT(&*ckvi, IsKeyValueScalars("kind", "DeclaredName"));
|
||||
++ckvi;
|
||||
EXPECT_THAT(&*ckvi, IsKeyValueScalars("text", "F"));
|
||||
++ckvi;
|
||||
|
||||
+257
-34
@@ -54,6 +54,38 @@ struct UnrecognizedDeclaration : SimpleDiagnostic<UnrecognizedDeclaration> {
|
||||
"Unrecognized declaration introducer.";
|
||||
};
|
||||
|
||||
struct ExpectedExpression : SimpleDiagnostic<ExpectedExpression> {
|
||||
static constexpr llvm::StringLiteral ShortName = "syntax-error";
|
||||
static constexpr llvm::StringLiteral Message = "Expected expression.";
|
||||
};
|
||||
|
||||
struct ExpectedCloseParen : SimpleDiagnostic<ExpectedCloseParen> {
|
||||
static constexpr llvm::StringLiteral ShortName = "syntax-error";
|
||||
static constexpr llvm::StringLiteral Message =
|
||||
"Unexpected tokens before `)`.";
|
||||
};
|
||||
|
||||
struct ExpectedSemiAfterExpression
|
||||
: SimpleDiagnostic<ExpectedSemiAfterExpression> {
|
||||
static constexpr llvm::StringLiteral ShortName = "syntax-error";
|
||||
static constexpr llvm::StringLiteral Message =
|
||||
"Expected `;` after expression.";
|
||||
};
|
||||
|
||||
struct ExpectedIdentifierAfterDot
|
||||
: SimpleDiagnostic<ExpectedIdentifierAfterDot> {
|
||||
static constexpr llvm::StringLiteral ShortName = "syntax-error";
|
||||
static constexpr llvm::StringLiteral Message =
|
||||
"Expected identifier after `.`.";
|
||||
};
|
||||
|
||||
struct UnexpectedTokenInFunctionArgs
|
||||
: SimpleDiagnostic<UnexpectedTokenInFunctionArgs> {
|
||||
static constexpr llvm::StringLiteral ShortName = "syntax-error";
|
||||
static constexpr llvm::StringLiteral Message =
|
||||
"Unexpected token in function argument list.";
|
||||
};
|
||||
|
||||
ParseTree::Parser::Parser(ParseTree& tree_arg, TokenizedBuffer& tokens_arg,
|
||||
TokenDiagnosticEmitter& emitter)
|
||||
: tree(tree_arg),
|
||||
@@ -130,16 +162,10 @@ auto ParseTree::Parser::MarkNodeError(Node n) -> void {
|
||||
|
||||
// A marker for the start of a node's subtree.
|
||||
//
|
||||
// This is used to track the size of the node's subtree and ensure at least one
|
||||
// parse node is added. It can be used repeatedly if multiple subtrees start at
|
||||
// the same position.
|
||||
// This is used to track the size of the node's subtree. It can be used
|
||||
// repeatedly if multiple subtrees start at the same position.
|
||||
struct ParseTree::Parser::SubtreeStart {
|
||||
int tree_size;
|
||||
bool node_added = false;
|
||||
|
||||
~SubtreeStart() {
|
||||
assert(node_added && "Never added a node for a subtree region!");
|
||||
}
|
||||
};
|
||||
|
||||
auto ParseTree::Parser::StartSubtree() -> SubtreeStart {
|
||||
@@ -147,7 +173,7 @@ auto ParseTree::Parser::StartSubtree() -> SubtreeStart {
|
||||
}
|
||||
|
||||
auto ParseTree::Parser::AddNode(ParseNodeKind n_kind, TokenizedBuffer::Token t,
|
||||
SubtreeStart& start, bool has_error) -> Node {
|
||||
SubtreeStart start, bool has_error) -> Node {
|
||||
// The size of the subtree is the change in size from when we started this
|
||||
// subtree to now, but including the node we're about to add.
|
||||
int tree_stop_size = static_cast<int>(tree.node_impls.size()) + 1;
|
||||
@@ -159,7 +185,6 @@ auto ParseTree::Parser::AddNode(ParseNodeKind n_kind, TokenizedBuffer::Token t,
|
||||
MarkNodeError(n);
|
||||
}
|
||||
|
||||
start.node_added = true;
|
||||
return n;
|
||||
}
|
||||
|
||||
@@ -181,11 +206,34 @@ auto ParseTree::Parser::SkipTo(TokenizedBuffer::Token t) -> void {
|
||||
assert(position != end && "Skipped past EOF.");
|
||||
}
|
||||
|
||||
auto ParseTree::Parser::SkipPastLikelyDeclarationEnd(
|
||||
TokenizedBuffer::Token skip_root, bool is_inside_declaration)
|
||||
auto ParseTree::Parser::FindNext(TokenKind desired_kind)
|
||||
-> llvm::Optional<TokenizedBuffer::Token> {
|
||||
auto new_position = position;
|
||||
while (true) {
|
||||
TokenizedBuffer::Token token = *new_position;
|
||||
TokenKind kind = tokens.GetKind(token);
|
||||
if (kind == desired_kind) {
|
||||
return token;
|
||||
}
|
||||
|
||||
// Step to the next token at the current bracketing level.
|
||||
if (kind.IsClosingSymbol() || kind == TokenKind::EndOfFile()) {
|
||||
// There are no more tokens at this level.
|
||||
return llvm::None;
|
||||
} else if (kind.IsOpeningSymbol()) {
|
||||
new_position =
|
||||
TokenizedBuffer::TokenIterator(tokens.GetMatchedClosingToken(token));
|
||||
} else {
|
||||
++new_position;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
auto ParseTree::Parser::SkipPastLikelyEnd(TokenizedBuffer::Token skip_root,
|
||||
SemiHandler on_semi)
|
||||
-> llvm::Optional<Node> {
|
||||
if (AtEndOfFile()) {
|
||||
return {};
|
||||
return llvm::None;
|
||||
}
|
||||
|
||||
TokenizedBuffer::Line root_line = tokens.GetLine(skip_root);
|
||||
@@ -208,17 +256,13 @@ auto ParseTree::Parser::SkipPastLikelyDeclarationEnd(
|
||||
if (current_kind == TokenKind::CloseCurlyBrace()) {
|
||||
// Immediately bail out if we hit an unmatched close curly, this will
|
||||
// pop us up a level of the syntax grouping.
|
||||
return {};
|
||||
return llvm::None;
|
||||
}
|
||||
|
||||
// If we find a semicolon, parse it and add a corresponding node. If we're
|
||||
// inside of a declaration, this is a declaration ending semicolon,
|
||||
// otherwise it simply forms an empty declaration.
|
||||
if (auto end_node = ConsumeAndAddLeafNodeIf(
|
||||
TokenKind::Semi(), is_inside_declaration
|
||||
? ParseNodeKind::DeclarationEnd()
|
||||
: ParseNodeKind::EmptyDeclaration())) {
|
||||
return end_node;
|
||||
// We assume that a semicolon is always intended to be the end of the
|
||||
// current construct.
|
||||
if (auto semi = ConsumeIf(TokenKind::Semi())) {
|
||||
return on_semi(*semi);
|
||||
}
|
||||
|
||||
// Skip over any matching group of tokens.
|
||||
@@ -231,7 +275,7 @@ auto ParseTree::Parser::SkipPastLikelyDeclarationEnd(
|
||||
} while (!AtEndOfFile() &&
|
||||
is_same_line_or_indent_greater_than_root(*position));
|
||||
|
||||
return {};
|
||||
return llvm::None;
|
||||
}
|
||||
|
||||
auto ParseTree::Parser::ParseFunctionSignature() -> Node {
|
||||
@@ -266,12 +310,18 @@ auto ParseTree::Parser::ParseCodeBlock() -> Node {
|
||||
for (;;) {
|
||||
switch (tokens.GetKind(*position)) {
|
||||
default:
|
||||
// FIXME: Add support for parsing more expressions & statements.
|
||||
emitter.EmitError<UnexpectedTokenInCodeBlock>(*position);
|
||||
has_errors = true;
|
||||
// A statement with no introducer token can only be an expression
|
||||
// statement.
|
||||
if (ParseExpressionStatement()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// We can trivially skip to the actual close curly brace from here.
|
||||
// We detected and diagnosed an error of some kind. We can trivially
|
||||
// skip to the actual close curly brace from here.
|
||||
// FIXME: It would be better to skip to the next semicolon, or the next
|
||||
// token at the start of a line with the same indent as this one.
|
||||
SkipTo(tokens.GetMatchedClosingToken(open_curly));
|
||||
has_errors = true;
|
||||
// Now fall through to the close curly brace handling code.
|
||||
LLVM_FALLTHROUGH;
|
||||
|
||||
@@ -306,21 +356,25 @@ auto ParseTree::Parser::ParseFunctionDeclaration() -> Node {
|
||||
start, /*has_error=*/true);
|
||||
};
|
||||
|
||||
auto handle_semi_in_error_recovery = [&](TokenizedBuffer::Token semi) {
|
||||
return AddLeafNode(ParseNodeKind::DeclarationEnd(), semi);
|
||||
};
|
||||
|
||||
auto name_n = ConsumeAndAddLeafNodeIf(TokenKind::Identifier(),
|
||||
ParseNodeKind::Identifier());
|
||||
ParseNodeKind::DeclaredName());
|
||||
if (!name_n) {
|
||||
emitter.EmitError<ExpectedFunctionName>(*position);
|
||||
// FIXME: We could change the lexer to allow us to synthesize certain
|
||||
// kinds of tokens and try to "recover" here, but unclear that this is
|
||||
// really useful.
|
||||
SkipPastLikelyDeclarationEnd(function_intro_token);
|
||||
SkipPastLikelyEnd(function_intro_token, handle_semi_in_error_recovery);
|
||||
return add_error_function_node();
|
||||
}
|
||||
|
||||
TokenizedBuffer::Token open_paren = *position;
|
||||
if (tokens.GetKind(open_paren) != TokenKind::OpenParen()) {
|
||||
emitter.EmitError<ExpectedFunctionParams>(open_paren);
|
||||
SkipPastLikelyDeclarationEnd(function_intro_token);
|
||||
SkipPastLikelyEnd(function_intro_token, handle_semi_in_error_recovery);
|
||||
return add_error_function_node();
|
||||
}
|
||||
TokenizedBuffer::Token close_paren =
|
||||
@@ -333,7 +387,7 @@ auto ParseTree::Parser::ParseFunctionDeclaration() -> Node {
|
||||
if (tree.node_impls[signature_n.index].has_error) {
|
||||
// Don't try to parse more of the function declaration, but consume a
|
||||
// declaration ending semicolon if found (without going to a new line).
|
||||
SkipPastLikelyDeclarationEnd(function_intro_token);
|
||||
SkipPastLikelyEnd(function_intro_token, handle_semi_in_error_recovery);
|
||||
return add_error_function_node();
|
||||
}
|
||||
|
||||
@@ -345,7 +399,7 @@ auto ParseTree::Parser::ParseFunctionDeclaration() -> Node {
|
||||
emitter.EmitError<ExpectedFunctionBodyOrSemi>(*position);
|
||||
if (tokens.GetLine(*position) == tokens.GetLine(close_paren)) {
|
||||
// Only need to skip if we've not already found a new line.
|
||||
SkipPastLikelyDeclarationEnd(function_intro_token);
|
||||
SkipPastLikelyEnd(function_intro_token, handle_semi_in_error_recovery);
|
||||
}
|
||||
return add_error_function_node();
|
||||
}
|
||||
@@ -380,7 +434,9 @@ auto ParseTree::Parser::ParseDeclaration() -> llvm::Optional<Node> {
|
||||
// Skip forward past any end of a declaration we simply didn't understand so
|
||||
// that we can find the start of the next declaration or the end of a scope.
|
||||
if (auto found_semi_n =
|
||||
SkipPastLikelyDeclarationEnd(t, /*is_inside_declaration=*/false)) {
|
||||
SkipPastLikelyEnd(t, [&](TokenizedBuffer::Token semi) {
|
||||
return AddLeafNode(ParseNodeKind::EmptyDeclaration(), semi);
|
||||
})) {
|
||||
MarkNodeError(*found_semi_n);
|
||||
return *found_semi_n;
|
||||
}
|
||||
@@ -388,7 +444,174 @@ auto ParseTree::Parser::ParseDeclaration() -> llvm::Optional<Node> {
|
||||
// Nothing, not even a semicolon found. We still need to mark that an error
|
||||
// occurred though.
|
||||
tree.has_errors = true;
|
||||
return {};
|
||||
return llvm::None;
|
||||
}
|
||||
|
||||
auto ParseTree::Parser::ParseParenExpression() -> llvm::Optional<Node> {
|
||||
// `(` expression `)`
|
||||
auto start = StartSubtree();
|
||||
TokenizedBuffer::Token open_paren = Consume(TokenKind::OpenParen());
|
||||
|
||||
// TODO: If the next token is a close paren, build an empty tuple literal.
|
||||
|
||||
bool has_errors = !ParseExpression();
|
||||
|
||||
// TODO: If the next token is a comma, build a tuple literal.
|
||||
|
||||
if (tokens.GetKind(*position) != TokenKind::CloseParen()) {
|
||||
if (!has_errors) {
|
||||
emitter.EmitError<ExpectedCloseParen>(*position);
|
||||
has_errors = true;
|
||||
}
|
||||
SkipTo(tokens.GetMatchedClosingToken(open_paren));
|
||||
}
|
||||
|
||||
AddLeafNode(ParseNodeKind::ParenExpressionEnd(),
|
||||
Consume(TokenKind::CloseParen()));
|
||||
return AddNode(ParseNodeKind::ParenExpression(), open_paren, start,
|
||||
has_errors);
|
||||
}
|
||||
|
||||
auto ParseTree::Parser::ParsePrimaryExpression() -> llvm::Optional<Node> {
|
||||
TokenizedBuffer::Token t = *position;
|
||||
TokenKind token_kind = tokens.GetKind(t);
|
||||
llvm::Optional<ParseNodeKind> kind;
|
||||
switch (token_kind) {
|
||||
case TokenKind::Identifier():
|
||||
kind = ParseNodeKind::NameReference();
|
||||
break;
|
||||
|
||||
case TokenKind::IntegerLiteral():
|
||||
case TokenKind::RealLiteral():
|
||||
case TokenKind::StringLiteral():
|
||||
kind = ParseNodeKind::Literal();
|
||||
break;
|
||||
|
||||
case TokenKind::OpenParen():
|
||||
return ParseParenExpression();
|
||||
|
||||
default:
|
||||
emitter.EmitError<ExpectedExpression>(t);
|
||||
return llvm::None;
|
||||
}
|
||||
|
||||
return AddLeafNode(*kind, Consume(token_kind));
|
||||
}
|
||||
|
||||
auto ParseTree::Parser::ParseDesignatorExpression(SubtreeStart start,
|
||||
bool has_errors)
|
||||
-> llvm::Optional<Node> {
|
||||
// `.` identifier
|
||||
auto dot = Consume(TokenKind::Period());
|
||||
auto name = ConsumeIf(TokenKind::Identifier());
|
||||
if (name) {
|
||||
AddLeafNode(ParseNodeKind::DesignatedName(), *name);
|
||||
} else {
|
||||
// If we see a keyword, assume it was intended to be the designated name.
|
||||
// TODO: Should keywords be valid in designators?
|
||||
if (tokens.GetKind(*position).IsKeyword()) {
|
||||
Consume(tokens.GetKind(*position));
|
||||
}
|
||||
emitter.EmitError<ExpectedIdentifierAfterDot>(*position);
|
||||
has_errors = true;
|
||||
}
|
||||
return AddNode(ParseNodeKind::DesignatorExpression(), dot, start, has_errors);
|
||||
}
|
||||
|
||||
auto ParseTree::Parser::ParseCallExpression(SubtreeStart start, bool has_errors)
|
||||
-> llvm::Optional<Node> {
|
||||
// `(` expression-list[opt] `)`
|
||||
//
|
||||
// expression-list ::= expression
|
||||
// ::= expression `,` expression-list
|
||||
TokenizedBuffer::Token open_paren = Consume(TokenKind::OpenParen());
|
||||
|
||||
// Parse arguments, if any are specified.
|
||||
if (tokens.GetKind(*position) != TokenKind::CloseParen()) {
|
||||
while (true) {
|
||||
bool argument_error = !ParseExpression();
|
||||
has_errors |= argument_error;
|
||||
|
||||
if (tokens.GetKind(*position) == TokenKind::CloseParen()) {
|
||||
break;
|
||||
}
|
||||
|
||||
if (tokens.GetKind(*position) != TokenKind::Comma()) {
|
||||
if (!argument_error) {
|
||||
emitter.EmitError<UnexpectedTokenInFunctionArgs>(*position);
|
||||
}
|
||||
has_errors = true;
|
||||
|
||||
auto comma_position = FindNext(TokenKind::Comma());
|
||||
if (!comma_position) {
|
||||
SkipTo(tokens.GetMatchedClosingToken(open_paren));
|
||||
break;
|
||||
}
|
||||
SkipTo(*comma_position);
|
||||
break;
|
||||
}
|
||||
|
||||
AddLeafNode(ParseNodeKind::CallExpressionComma(),
|
||||
Consume(TokenKind::Comma()));
|
||||
}
|
||||
}
|
||||
|
||||
AddLeafNode(ParseNodeKind::CallExpressionEnd(),
|
||||
Consume(TokenKind::CloseParen()));
|
||||
return AddNode(ParseNodeKind::CallExpression(), open_paren, start,
|
||||
has_errors);
|
||||
}
|
||||
|
||||
auto ParseTree::Parser::ParsePostfixExpression() -> llvm::Optional<Node> {
|
||||
auto start = StartSubtree();
|
||||
llvm::Optional<Node> expression = ParsePrimaryExpression();
|
||||
|
||||
while (true) {
|
||||
switch (tokens.GetKind(*position)) {
|
||||
case TokenKind::Period():
|
||||
expression = ParseDesignatorExpression(start, !expression);
|
||||
break;
|
||||
|
||||
case TokenKind::OpenParen():
|
||||
expression = ParseCallExpression(start, !expression);
|
||||
break;
|
||||
|
||||
default: {
|
||||
return expression;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
auto ParseTree::Parser::ParseExpression() -> llvm::Optional<Node> {
|
||||
return ParsePostfixExpression();
|
||||
}
|
||||
|
||||
auto ParseTree::Parser::ParseExpressionStatement() -> llvm::Optional<Node> {
|
||||
TokenizedBuffer::Token start_token = *position;
|
||||
auto start = StartSubtree();
|
||||
|
||||
bool has_errors = !ParseExpression();
|
||||
|
||||
if (auto semi = ConsumeIf(TokenKind::Semi())) {
|
||||
return AddNode(ParseNodeKind::ExpressionStatement(), *semi, start,
|
||||
has_errors);
|
||||
}
|
||||
|
||||
if (!has_errors) {
|
||||
emitter.EmitError<ExpectedSemiAfterExpression>(*position);
|
||||
}
|
||||
|
||||
if (auto recovery_node =
|
||||
SkipPastLikelyEnd(start_token, [&](TokenizedBuffer::Token semi) {
|
||||
return AddNode(ParseNodeKind::ExpressionStatement(), semi, start,
|
||||
true);
|
||||
})) {
|
||||
return recovery_node;
|
||||
}
|
||||
|
||||
// Found junk not even followed by a `;`.
|
||||
return llvm::None;
|
||||
}
|
||||
|
||||
} // namespace Carbon
|
||||
|
||||
+53
-19
@@ -58,9 +58,8 @@ class ParseTree::Parser {
|
||||
|
||||
// Start parsing one (or more) subtrees of nodes.
|
||||
//
|
||||
// This returns a marker representing start position. It will also enforce
|
||||
// that at least *some* node is added using this starting position. Multiple
|
||||
// nodes can be added if they share a start position though.
|
||||
// This returns a marker representing start position. Multiple nodes can be
|
||||
// added if they share a start position.
|
||||
auto StartSubtree() -> SubtreeStart;
|
||||
|
||||
// Add a node to the parse tree that potentially has a subtree larger than
|
||||
@@ -69,7 +68,7 @@ class ParseTree::Parser {
|
||||
// Requires a start marker be passed to compute the size of the subtree rooted
|
||||
// at this node.
|
||||
auto AddNode(ParseNodeKind n_kind, TokenizedBuffer::Token t,
|
||||
SubtreeStart& start, bool has_error = false) -> Node;
|
||||
SubtreeStart start, bool has_error = false) -> Node;
|
||||
|
||||
// If the current token is an opening symbol for a matched group, skips
|
||||
// forward to one past the matched closing symbol and returns true. Otherwise,
|
||||
@@ -79,28 +78,33 @@ class ParseTree::Parser {
|
||||
// Skip forward to the token immediately after the given token.
|
||||
auto SkipTo(TokenizedBuffer::Token t) -> void;
|
||||
|
||||
// Skips forward to move past the likely end of a declaration.
|
||||
// Find the next token of the given kind at the current bracketing level.
|
||||
auto FindNext(TokenKind t) -> llvm::Optional<TokenizedBuffer::Token>;
|
||||
|
||||
// Callback used if we find a semicolon when skipping to the end of a
|
||||
// declaration or statement.
|
||||
using SemiHandler = llvm::function_ref<
|
||||
auto(TokenizedBuffer::Token semi)->llvm::Optional<Node>>;
|
||||
|
||||
// Skips forward to move past the likely end of a declaration or statement.
|
||||
//
|
||||
// Looks forward, skipping over any matched symbol groups, to find the next
|
||||
// position that is likely past the end of a declaration. This is a heuristic
|
||||
// and should only be called when skipping past parse errors.
|
||||
// position that is likely past the end of a declaration or statement. This
|
||||
// is a heuristic and should only be called when skipping past parse errors.
|
||||
//
|
||||
// The strategy for recognizing when we have likely passed the end of a
|
||||
// declaration:
|
||||
// - If we get to close curly brace, we likely ended the entire context of
|
||||
// declarations.
|
||||
// - If we get to a semicolon, that should have ended the declaration.
|
||||
// declaration or statement:
|
||||
// - If we get to close curly brace, we likely ended the entire context.
|
||||
// - If we get to a semicolon, that should have ended the declaration or
|
||||
// statement.
|
||||
// - If we get to a new line from the `SkipRoot` token, but with the same or
|
||||
// less indentation, there is likely a missing semicolon. Continued
|
||||
// declarations across multiple lines should be indented.
|
||||
// declarations or statements across multiple lines should be indented.
|
||||
//
|
||||
// If we find a semicolon based on this skipping, we try to build a parse node
|
||||
// to represent it and will return that node. Otherwise we will return an
|
||||
// empty optional. If `IsInsideDeclaration` is true (the default) we build a
|
||||
// node that marks the end of the declaration we are inside. Otherwise we
|
||||
// build an empty declaration node.
|
||||
auto SkipPastLikelyDeclarationEnd(TokenizedBuffer::Token skip_root,
|
||||
bool is_inside_declaration = true)
|
||||
// If we find a semicolon based on this skipping, we call `on_semi_` to try
|
||||
// to build a parse node to represent it, and will return that node.
|
||||
// Otherwise we will return an empty optional.
|
||||
auto SkipPastLikelyEnd(TokenizedBuffer::Token skip_root, SemiHandler on_semi)
|
||||
-> llvm::Optional<Node>;
|
||||
|
||||
// Parses the signature of the function, consisting of a parameter list and an
|
||||
@@ -126,6 +130,36 @@ class ParseTree::Parser {
|
||||
// even when a node is returned.
|
||||
auto ParseDeclaration() -> llvm::Optional<Node>;
|
||||
|
||||
// Parses a parenthesized expression.
|
||||
auto ParseParenExpression() -> llvm::Optional<Node>;
|
||||
|
||||
// Parses a primary expression, which is either a terminal portion of an
|
||||
// expression tree, such as an identifier or literal, or a parenthesized
|
||||
// expression.
|
||||
auto ParsePrimaryExpression() -> llvm::Optional<Node>;
|
||||
|
||||
// Parses a designator expression suffix starting with `.`.
|
||||
auto ParseDesignatorExpression(SubtreeStart start, bool has_errors)
|
||||
-> llvm::Optional<Node>;
|
||||
|
||||
// Parses a call expression suffix starting with `(`.
|
||||
auto ParseCallExpression(SubtreeStart start, bool has_errors)
|
||||
-> llvm::Optional<Node>;
|
||||
|
||||
// Parses a postfix expression, which is a primary expression followed by
|
||||
// zero or more of the following:
|
||||
//
|
||||
// - function applications
|
||||
// - array indexes (TODO)
|
||||
// - designators
|
||||
auto ParsePostfixExpression() -> llvm::Optional<Node>;
|
||||
|
||||
// Parses an expression.
|
||||
auto ParseExpression() -> llvm::Optional<Node>;
|
||||
|
||||
// Parses an expression statement: an expression followed by a semicolon.
|
||||
auto ParseExpressionStatement() -> llvm::Optional<Node>;
|
||||
|
||||
ParseTree& tree;
|
||||
TokenizedBuffer& tokens;
|
||||
TokenDiagnosticEmitter& emitter;
|
||||
|
||||
Reference in New Issue
Block a user