diff --git a/parser/parse_node_kind.def b/parser/parse_node_kind.def index c605622f6798..8a4985ad537b 100644 --- a/parser/parse_node_kind.def +++ b/parser/parse_node_kind.def @@ -13,14 +13,29 @@ #error "Must define the x-macro to use this file." #endif -CARBON_PARSE_NODE_KIND(CodeBlockEnd) -CARBON_PARSE_NODE_KIND(CodeBlock) +// Declarations. CARBON_PARSE_NODE_KIND(DeclarationEnd) CARBON_PARSE_NODE_KIND(EmptyDeclaration) CARBON_PARSE_NODE_KIND(FunctionDeclaration) -CARBON_PARSE_NODE_KIND(Identifier) +CARBON_PARSE_NODE_KIND(DeclaredName) CARBON_PARSE_NODE_KIND(ParameterListEnd) CARBON_PARSE_NODE_KIND(ParameterList) CARBON_PARSE_NODE_KIND(FileEnd) +// Statements. +CARBON_PARSE_NODE_KIND(CodeBlockEnd) +CARBON_PARSE_NODE_KIND(CodeBlock) +CARBON_PARSE_NODE_KIND(ExpressionStatement) + +// Expressions. +CARBON_PARSE_NODE_KIND(Literal) +CARBON_PARSE_NODE_KIND(NameReference) +CARBON_PARSE_NODE_KIND(ParenExpression) +CARBON_PARSE_NODE_KIND(ParenExpressionEnd) +CARBON_PARSE_NODE_KIND(DesignatorExpression) +CARBON_PARSE_NODE_KIND(DesignatedName) +CARBON_PARSE_NODE_KIND(CallExpression) +CARBON_PARSE_NODE_KIND(CallExpressionComma) +CARBON_PARSE_NODE_KIND(CallExpressionEnd) + #undef CARBON_PARSE_NODE_KIND diff --git a/parser/parse_tree_test.cpp b/parser/parse_tree_test.cpp index cb2949300a23..fbda71a7835f 100644 --- a/parser/parse_tree_test.cpp +++ b/parser/parse_tree_test.cpp @@ -20,6 +20,7 @@ namespace Carbon { namespace { +using Carbon::Testing::ExpectedNode; using Carbon::Testing::IsKeyValueScalars; using Carbon::Testing::MatchParseTreeNodes; using namespace Carbon::Testing::NodeMatchers; @@ -91,7 +92,7 @@ TEST_F(ParseTreeTest, BasicFunctionDeclaration) { EXPECT_THAT(tree, MatchParseTreeNodes( {MatchFunctionDeclaration( - "fn", MatchIdentifier("F"), + "fn", MatchDeclaredName("F"), MatchParameterList("(", MatchParameterListEnd(")")), MatchDeclarationEnd(";")), MatchFileEnd()})); @@ -134,9 +135,10 @@ TEST_F(ParseTreeTest, FunctionDeclarationWithNoSignatureOrSemi) { TokenizedBuffer tokens = GetTokenizedBuffer("fn foo"); ParseTree tree = ParseTree::Parse(tokens, consumer); EXPECT_TRUE(tree.HasErrors()); - EXPECT_THAT(tree, MatchParseTreeNodes({MatchFunctionDeclaration( - HasError, MatchIdentifier("foo")), - MatchFileEnd()})); + EXPECT_THAT(tree, + MatchParseTreeNodes( + {MatchFunctionDeclaration(HasError, MatchDeclaredName("foo")), + MatchFileEnd()})); } TEST_F(ParseTreeTest, @@ -145,7 +147,7 @@ TEST_F(ParseTreeTest, ParseTree tree = ParseTree::Parse(tokens, consumer); EXPECT_TRUE(tree.HasErrors()); EXPECT_THAT(tree, MatchParseTreeNodes({MatchFunctionDeclaration( - HasError, MatchIdentifier("foo"), + HasError, MatchDeclaredName("foo"), MatchDeclarationEnd()), MatchFileEnd()})); } @@ -159,7 +161,7 @@ TEST_F(ParseTreeTest, FunctionDeclarationWithSingleIdentifierParameterList) { EXPECT_THAT(tree, MatchParseTreeNodes( {MatchFunctionDeclaration( - HasError, MatchIdentifier("foo"), + HasError, MatchDeclaredName("foo"), MatchParameterList(HasError, MatchParameterListEnd()), MatchDeclarationEnd()), MatchFileEnd()})); @@ -195,7 +197,7 @@ TEST_F(ParseTreeTest, FunctionDeclarationSkipToNewlineWithoutSemi) { tree, MatchParseTreeNodes( {MatchFunctionDeclaration(HasError), - MatchFunctionDeclaration(MatchIdentifier("F"), + MatchFunctionDeclaration(MatchDeclaredName("F"), MatchParameterList(MatchParameterListEnd()), MatchDeclarationEnd()), MatchFileEnd()})); @@ -213,7 +215,7 @@ TEST_F(ParseTreeTest, FunctionDeclarationSkipIndentedNewlineWithSemi) { tree, MatchParseTreeNodes( {MatchFunctionDeclaration(HasError, MatchDeclarationEnd()), - MatchFunctionDeclaration(MatchIdentifier("F"), + MatchFunctionDeclaration(MatchDeclaredName("F"), MatchParameterList(MatchParameterListEnd()), MatchDeclarationEnd()), MatchFileEnd()})); @@ -231,7 +233,7 @@ TEST_F(ParseTreeTest, FunctionDeclarationSkipIndentedNewlineWithoutSemi) { tree, MatchParseTreeNodes( {MatchFunctionDeclaration(HasError), - MatchFunctionDeclaration(MatchIdentifier("F"), + MatchFunctionDeclaration(MatchDeclaredName("F"), MatchParameterList(MatchParameterListEnd()), MatchDeclarationEnd()), MatchFileEnd()})); @@ -249,7 +251,7 @@ TEST_F(ParseTreeTest, FunctionDeclarationSkipIndentedNewlineUntilOutdent) { tree, MatchParseTreeNodes( {MatchFunctionDeclaration(HasError), - MatchFunctionDeclaration(MatchIdentifier("F"), + MatchFunctionDeclaration(MatchDeclaredName("F"), MatchParameterList(MatchParameterListEnd()), MatchDeclarationEnd()), MatchFileEnd()})); @@ -276,7 +278,7 @@ TEST_F(ParseTreeTest, BasicFunctionDefinition) { EXPECT_FALSE(tree.HasErrors()); EXPECT_THAT(tree, MatchParseTreeNodes( {MatchFunctionDeclaration( - MatchIdentifier("F"), + MatchDeclaredName("F"), MatchParameterList(MatchParameterListEnd()), MatchCodeBlock("{", MatchCodeBlockEnd("}"))), MatchFileEnd()})); @@ -294,7 +296,7 @@ TEST_F(ParseTreeTest, FunctionDefinitionWithNestedBlocks) { EXPECT_THAT( tree, MatchParseTreeNodes( {MatchFunctionDeclaration( - MatchIdentifier("F"), + MatchDeclaredName("F"), MatchParameterList(MatchParameterListEnd()), MatchCodeBlock( MatchCodeBlock( @@ -316,9 +318,10 @@ TEST_F(ParseTreeTest, FunctionDefinitionWithIdenifierInStatements) { EXPECT_TRUE(tree.HasErrors()); EXPECT_THAT(tree, MatchParseTreeNodes( {MatchFunctionDeclaration( - MatchIdentifier("F"), + MatchDeclaredName("F"), MatchParameterList(MatchParameterListEnd()), - MatchCodeBlock(HasError, MatchCodeBlockEnd())), + MatchCodeBlock(HasError, MatchNameReference("bar"), + MatchCodeBlockEnd())), MatchFileEnd()})); } @@ -331,13 +334,76 @@ TEST_F(ParseTreeTest, FunctionDefinitionWithIdenifierInNestedBlock) { // Note: this might become valid depending on the expression syntax. This test // shouldn't be taken as a sign it should remain invalid. EXPECT_TRUE(tree.HasErrors()); + EXPECT_THAT(tree, + MatchParseTreeNodes( + {MatchFunctionDeclaration( + MatchDeclaredName("F"), + MatchParameterList(MatchParameterListEnd()), + MatchCodeBlock( + MatchCodeBlock(HasError, MatchNameReference("bar"), + MatchCodeBlockEnd()), + MatchCodeBlockEnd())), + MatchFileEnd()})); +} + +TEST_F(ParseTreeTest, FunctionDefinitionWithFunctionCall) { + TokenizedBuffer tokens = GetTokenizedBuffer( + "fn F() {\n" + " a.b.f(c.d, (e)).g();\n" + "}"); + ParseTree tree = ParseTree::Parse(tokens, consumer); + EXPECT_FALSE(tree.HasErrors()); + + ExpectedNode call_to_f = MatchCallExpression( + MatchDesignatorExpression( + MatchDesignatorExpression(MatchNameReference("a"), + MatchDesignatedName("b")), + MatchDesignatedName("f")), + MatchDesignatorExpression(MatchNameReference("c"), + MatchDesignatedName("d")), + MatchCallExpressionComma(), + MatchParenExpression(MatchNameReference("e"), MatchParenExpressionEnd()), + MatchCallExpressionEnd()); + ExpectedNode statement = MatchExpressionStatement(MatchCallExpression( + MatchDesignatorExpression(call_to_f, MatchDesignatedName("g")), + MatchCallExpressionEnd())); + + EXPECT_THAT(tree, MatchParseTreeNodes( + {MatchFunctionDeclaration( + MatchDeclaredName("F"), + MatchParameterList(MatchParameterListEnd()), + MatchCodeBlock(statement, MatchCodeBlockEnd())), + MatchFileEnd()})); +} + +TEST_F(ParseTreeTest, InvalidDesignators) { + TokenizedBuffer tokens = GetTokenizedBuffer( + "fn F() {\n" + " a.;\n" + " a.fn;\n" + " a.42;\n" + "}"); + ParseTree tree = ParseTree::Parse(tokens, consumer); + EXPECT_TRUE(tree.HasErrors()); + EXPECT_THAT( tree, MatchParseTreeNodes( {MatchFunctionDeclaration( - MatchIdentifier("F"), + MatchDeclaredName("F"), MatchParameterList(MatchParameterListEnd()), - MatchCodeBlock(MatchCodeBlock(HasError, MatchCodeBlockEnd()), + MatchCodeBlock(MatchExpressionStatement( + MatchDesignatorExpression( + MatchNameReference("a"), ".", HasError), + ";"), + MatchExpressionStatement( + MatchDesignatorExpression( + MatchNameReference("a"), ".", HasError), + ";"), + MatchExpressionStatement( + MatchDesignatorExpression( + MatchNameReference("a"), ".", HasError), + HasError, ";"), MatchCodeBlockEnd())), MatchFileEnd()})); } @@ -368,7 +434,7 @@ TEST_F(ParseTreeTest, Printing) { StrEq("{node_index: 4, kind: 'FunctionDeclaration', text: 'fn', " "subtree_size: 5, children: [")); EXPECT_THAT(GetAndDropLine(print), - StrEq(" {node_index: 0, kind: 'Identifier', text: 'F'},")); + StrEq(" {node_index: 0, kind: 'DeclaredName', text: 'F'},")); EXPECT_THAT(GetAndDropLine(print), StrEq(" {node_index: 2, kind: 'ParameterList', text: '(', " "subtree_size: 2, children: [")); @@ -432,7 +498,7 @@ TEST_F(ParseTreeTest, PrintingAsYAML) { auto ckvi = node->begin(); EXPECT_THAT(&*ckvi, IsKeyValueScalars("node_index", "0")); ++ckvi; - EXPECT_THAT(&*ckvi, IsKeyValueScalars("kind", "Identifier")); + EXPECT_THAT(&*ckvi, IsKeyValueScalars("kind", "DeclaredName")); ++ckvi; EXPECT_THAT(&*ckvi, IsKeyValueScalars("text", "F")); ++ckvi; diff --git a/parser/parser_impl.cpp b/parser/parser_impl.cpp index e3b29b54cc3e..ba32635b6910 100644 --- a/parser/parser_impl.cpp +++ b/parser/parser_impl.cpp @@ -54,6 +54,38 @@ struct UnrecognizedDeclaration : SimpleDiagnostic { "Unrecognized declaration introducer."; }; +struct ExpectedExpression : SimpleDiagnostic { + static constexpr llvm::StringLiteral ShortName = "syntax-error"; + static constexpr llvm::StringLiteral Message = "Expected expression."; +}; + +struct ExpectedCloseParen : SimpleDiagnostic { + static constexpr llvm::StringLiteral ShortName = "syntax-error"; + static constexpr llvm::StringLiteral Message = + "Unexpected tokens before `)`."; +}; + +struct ExpectedSemiAfterExpression + : SimpleDiagnostic { + static constexpr llvm::StringLiteral ShortName = "syntax-error"; + static constexpr llvm::StringLiteral Message = + "Expected `;` after expression."; +}; + +struct ExpectedIdentifierAfterDot + : SimpleDiagnostic { + static constexpr llvm::StringLiteral ShortName = "syntax-error"; + static constexpr llvm::StringLiteral Message = + "Expected identifier after `.`."; +}; + +struct UnexpectedTokenInFunctionArgs + : SimpleDiagnostic { + static constexpr llvm::StringLiteral ShortName = "syntax-error"; + static constexpr llvm::StringLiteral Message = + "Unexpected token in function argument list."; +}; + ParseTree::Parser::Parser(ParseTree& tree_arg, TokenizedBuffer& tokens_arg, TokenDiagnosticEmitter& emitter) : tree(tree_arg), @@ -130,16 +162,10 @@ auto ParseTree::Parser::MarkNodeError(Node n) -> void { // A marker for the start of a node's subtree. // -// This is used to track the size of the node's subtree and ensure at least one -// parse node is added. It can be used repeatedly if multiple subtrees start at -// the same position. +// This is used to track the size of the node's subtree. It can be used +// repeatedly if multiple subtrees start at the same position. struct ParseTree::Parser::SubtreeStart { int tree_size; - bool node_added = false; - - ~SubtreeStart() { - assert(node_added && "Never added a node for a subtree region!"); - } }; auto ParseTree::Parser::StartSubtree() -> SubtreeStart { @@ -147,7 +173,7 @@ auto ParseTree::Parser::StartSubtree() -> SubtreeStart { } auto ParseTree::Parser::AddNode(ParseNodeKind n_kind, TokenizedBuffer::Token t, - SubtreeStart& start, bool has_error) -> Node { + SubtreeStart start, bool has_error) -> Node { // The size of the subtree is the change in size from when we started this // subtree to now, but including the node we're about to add. int tree_stop_size = static_cast(tree.node_impls.size()) + 1; @@ -159,7 +185,6 @@ auto ParseTree::Parser::AddNode(ParseNodeKind n_kind, TokenizedBuffer::Token t, MarkNodeError(n); } - start.node_added = true; return n; } @@ -181,11 +206,34 @@ auto ParseTree::Parser::SkipTo(TokenizedBuffer::Token t) -> void { assert(position != end && "Skipped past EOF."); } -auto ParseTree::Parser::SkipPastLikelyDeclarationEnd( - TokenizedBuffer::Token skip_root, bool is_inside_declaration) +auto ParseTree::Parser::FindNext(TokenKind desired_kind) + -> llvm::Optional { + auto new_position = position; + while (true) { + TokenizedBuffer::Token token = *new_position; + TokenKind kind = tokens.GetKind(token); + if (kind == desired_kind) { + return token; + } + + // Step to the next token at the current bracketing level. + if (kind.IsClosingSymbol() || kind == TokenKind::EndOfFile()) { + // There are no more tokens at this level. + return llvm::None; + } else if (kind.IsOpeningSymbol()) { + new_position = + TokenizedBuffer::TokenIterator(tokens.GetMatchedClosingToken(token)); + } else { + ++new_position; + } + } +} + +auto ParseTree::Parser::SkipPastLikelyEnd(TokenizedBuffer::Token skip_root, + SemiHandler on_semi) -> llvm::Optional { if (AtEndOfFile()) { - return {}; + return llvm::None; } TokenizedBuffer::Line root_line = tokens.GetLine(skip_root); @@ -208,17 +256,13 @@ auto ParseTree::Parser::SkipPastLikelyDeclarationEnd( if (current_kind == TokenKind::CloseCurlyBrace()) { // Immediately bail out if we hit an unmatched close curly, this will // pop us up a level of the syntax grouping. - return {}; + return llvm::None; } - // If we find a semicolon, parse it and add a corresponding node. If we're - // inside of a declaration, this is a declaration ending semicolon, - // otherwise it simply forms an empty declaration. - if (auto end_node = ConsumeAndAddLeafNodeIf( - TokenKind::Semi(), is_inside_declaration - ? ParseNodeKind::DeclarationEnd() - : ParseNodeKind::EmptyDeclaration())) { - return end_node; + // We assume that a semicolon is always intended to be the end of the + // current construct. + if (auto semi = ConsumeIf(TokenKind::Semi())) { + return on_semi(*semi); } // Skip over any matching group of tokens. @@ -231,7 +275,7 @@ auto ParseTree::Parser::SkipPastLikelyDeclarationEnd( } while (!AtEndOfFile() && is_same_line_or_indent_greater_than_root(*position)); - return {}; + return llvm::None; } auto ParseTree::Parser::ParseFunctionSignature() -> Node { @@ -266,12 +310,18 @@ auto ParseTree::Parser::ParseCodeBlock() -> Node { for (;;) { switch (tokens.GetKind(*position)) { default: - // FIXME: Add support for parsing more expressions & statements. - emitter.EmitError(*position); - has_errors = true; + // A statement with no introducer token can only be an expression + // statement. + if (ParseExpressionStatement()) { + continue; + } - // We can trivially skip to the actual close curly brace from here. + // We detected and diagnosed an error of some kind. We can trivially + // skip to the actual close curly brace from here. + // FIXME: It would be better to skip to the next semicolon, or the next + // token at the start of a line with the same indent as this one. SkipTo(tokens.GetMatchedClosingToken(open_curly)); + has_errors = true; // Now fall through to the close curly brace handling code. LLVM_FALLTHROUGH; @@ -306,21 +356,25 @@ auto ParseTree::Parser::ParseFunctionDeclaration() -> Node { start, /*has_error=*/true); }; + auto handle_semi_in_error_recovery = [&](TokenizedBuffer::Token semi) { + return AddLeafNode(ParseNodeKind::DeclarationEnd(), semi); + }; + auto name_n = ConsumeAndAddLeafNodeIf(TokenKind::Identifier(), - ParseNodeKind::Identifier()); + ParseNodeKind::DeclaredName()); if (!name_n) { emitter.EmitError(*position); // FIXME: We could change the lexer to allow us to synthesize certain // kinds of tokens and try to "recover" here, but unclear that this is // really useful. - SkipPastLikelyDeclarationEnd(function_intro_token); + SkipPastLikelyEnd(function_intro_token, handle_semi_in_error_recovery); return add_error_function_node(); } TokenizedBuffer::Token open_paren = *position; if (tokens.GetKind(open_paren) != TokenKind::OpenParen()) { emitter.EmitError(open_paren); - SkipPastLikelyDeclarationEnd(function_intro_token); + SkipPastLikelyEnd(function_intro_token, handle_semi_in_error_recovery); return add_error_function_node(); } TokenizedBuffer::Token close_paren = @@ -333,7 +387,7 @@ auto ParseTree::Parser::ParseFunctionDeclaration() -> Node { if (tree.node_impls[signature_n.index].has_error) { // Don't try to parse more of the function declaration, but consume a // declaration ending semicolon if found (without going to a new line). - SkipPastLikelyDeclarationEnd(function_intro_token); + SkipPastLikelyEnd(function_intro_token, handle_semi_in_error_recovery); return add_error_function_node(); } @@ -345,7 +399,7 @@ auto ParseTree::Parser::ParseFunctionDeclaration() -> Node { emitter.EmitError(*position); if (tokens.GetLine(*position) == tokens.GetLine(close_paren)) { // Only need to skip if we've not already found a new line. - SkipPastLikelyDeclarationEnd(function_intro_token); + SkipPastLikelyEnd(function_intro_token, handle_semi_in_error_recovery); } return add_error_function_node(); } @@ -380,7 +434,9 @@ auto ParseTree::Parser::ParseDeclaration() -> llvm::Optional { // Skip forward past any end of a declaration we simply didn't understand so // that we can find the start of the next declaration or the end of a scope. if (auto found_semi_n = - SkipPastLikelyDeclarationEnd(t, /*is_inside_declaration=*/false)) { + SkipPastLikelyEnd(t, [&](TokenizedBuffer::Token semi) { + return AddLeafNode(ParseNodeKind::EmptyDeclaration(), semi); + })) { MarkNodeError(*found_semi_n); return *found_semi_n; } @@ -388,7 +444,174 @@ auto ParseTree::Parser::ParseDeclaration() -> llvm::Optional { // Nothing, not even a semicolon found. We still need to mark that an error // occurred though. tree.has_errors = true; - return {}; + return llvm::None; +} + +auto ParseTree::Parser::ParseParenExpression() -> llvm::Optional { + // `(` expression `)` + auto start = StartSubtree(); + TokenizedBuffer::Token open_paren = Consume(TokenKind::OpenParen()); + + // TODO: If the next token is a close paren, build an empty tuple literal. + + bool has_errors = !ParseExpression(); + + // TODO: If the next token is a comma, build a tuple literal. + + if (tokens.GetKind(*position) != TokenKind::CloseParen()) { + if (!has_errors) { + emitter.EmitError(*position); + has_errors = true; + } + SkipTo(tokens.GetMatchedClosingToken(open_paren)); + } + + AddLeafNode(ParseNodeKind::ParenExpressionEnd(), + Consume(TokenKind::CloseParen())); + return AddNode(ParseNodeKind::ParenExpression(), open_paren, start, + has_errors); +} + +auto ParseTree::Parser::ParsePrimaryExpression() -> llvm::Optional { + TokenizedBuffer::Token t = *position; + TokenKind token_kind = tokens.GetKind(t); + llvm::Optional kind; + switch (token_kind) { + case TokenKind::Identifier(): + kind = ParseNodeKind::NameReference(); + break; + + case TokenKind::IntegerLiteral(): + case TokenKind::RealLiteral(): + case TokenKind::StringLiteral(): + kind = ParseNodeKind::Literal(); + break; + + case TokenKind::OpenParen(): + return ParseParenExpression(); + + default: + emitter.EmitError(t); + return llvm::None; + } + + return AddLeafNode(*kind, Consume(token_kind)); +} + +auto ParseTree::Parser::ParseDesignatorExpression(SubtreeStart start, + bool has_errors) + -> llvm::Optional { + // `.` identifier + auto dot = Consume(TokenKind::Period()); + auto name = ConsumeIf(TokenKind::Identifier()); + if (name) { + AddLeafNode(ParseNodeKind::DesignatedName(), *name); + } else { + // If we see a keyword, assume it was intended to be the designated name. + // TODO: Should keywords be valid in designators? + if (tokens.GetKind(*position).IsKeyword()) { + Consume(tokens.GetKind(*position)); + } + emitter.EmitError(*position); + has_errors = true; + } + return AddNode(ParseNodeKind::DesignatorExpression(), dot, start, has_errors); +} + +auto ParseTree::Parser::ParseCallExpression(SubtreeStart start, bool has_errors) + -> llvm::Optional { + // `(` expression-list[opt] `)` + // + // expression-list ::= expression + // ::= expression `,` expression-list + TokenizedBuffer::Token open_paren = Consume(TokenKind::OpenParen()); + + // Parse arguments, if any are specified. + if (tokens.GetKind(*position) != TokenKind::CloseParen()) { + while (true) { + bool argument_error = !ParseExpression(); + has_errors |= argument_error; + + if (tokens.GetKind(*position) == TokenKind::CloseParen()) { + break; + } + + if (tokens.GetKind(*position) != TokenKind::Comma()) { + if (!argument_error) { + emitter.EmitError(*position); + } + has_errors = true; + + auto comma_position = FindNext(TokenKind::Comma()); + if (!comma_position) { + SkipTo(tokens.GetMatchedClosingToken(open_paren)); + break; + } + SkipTo(*comma_position); + break; + } + + AddLeafNode(ParseNodeKind::CallExpressionComma(), + Consume(TokenKind::Comma())); + } + } + + AddLeafNode(ParseNodeKind::CallExpressionEnd(), + Consume(TokenKind::CloseParen())); + return AddNode(ParseNodeKind::CallExpression(), open_paren, start, + has_errors); +} + +auto ParseTree::Parser::ParsePostfixExpression() -> llvm::Optional { + auto start = StartSubtree(); + llvm::Optional expression = ParsePrimaryExpression(); + + while (true) { + switch (tokens.GetKind(*position)) { + case TokenKind::Period(): + expression = ParseDesignatorExpression(start, !expression); + break; + + case TokenKind::OpenParen(): + expression = ParseCallExpression(start, !expression); + break; + + default: { + return expression; + } + } + } +} + +auto ParseTree::Parser::ParseExpression() -> llvm::Optional { + return ParsePostfixExpression(); +} + +auto ParseTree::Parser::ParseExpressionStatement() -> llvm::Optional { + TokenizedBuffer::Token start_token = *position; + auto start = StartSubtree(); + + bool has_errors = !ParseExpression(); + + if (auto semi = ConsumeIf(TokenKind::Semi())) { + return AddNode(ParseNodeKind::ExpressionStatement(), *semi, start, + has_errors); + } + + if (!has_errors) { + emitter.EmitError(*position); + } + + if (auto recovery_node = + SkipPastLikelyEnd(start_token, [&](TokenizedBuffer::Token semi) { + return AddNode(ParseNodeKind::ExpressionStatement(), semi, start, + true); + })) { + return recovery_node; + } + + // Found junk not even followed by a `;`. + return llvm::None; } } // namespace Carbon diff --git a/parser/parser_impl.h b/parser/parser_impl.h index 52f9bbffa0cc..11c15cd55094 100644 --- a/parser/parser_impl.h +++ b/parser/parser_impl.h @@ -58,9 +58,8 @@ class ParseTree::Parser { // Start parsing one (or more) subtrees of nodes. // - // This returns a marker representing start position. It will also enforce - // that at least *some* node is added using this starting position. Multiple - // nodes can be added if they share a start position though. + // This returns a marker representing start position. Multiple nodes can be + // added if they share a start position. auto StartSubtree() -> SubtreeStart; // Add a node to the parse tree that potentially has a subtree larger than @@ -69,7 +68,7 @@ class ParseTree::Parser { // Requires a start marker be passed to compute the size of the subtree rooted // at this node. auto AddNode(ParseNodeKind n_kind, TokenizedBuffer::Token t, - SubtreeStart& start, bool has_error = false) -> Node; + SubtreeStart start, bool has_error = false) -> Node; // If the current token is an opening symbol for a matched group, skips // forward to one past the matched closing symbol and returns true. Otherwise, @@ -79,28 +78,33 @@ class ParseTree::Parser { // Skip forward to the token immediately after the given token. auto SkipTo(TokenizedBuffer::Token t) -> void; - // Skips forward to move past the likely end of a declaration. + // Find the next token of the given kind at the current bracketing level. + auto FindNext(TokenKind t) -> llvm::Optional; + + // Callback used if we find a semicolon when skipping to the end of a + // declaration or statement. + using SemiHandler = llvm::function_ref< + auto(TokenizedBuffer::Token semi)->llvm::Optional>; + + // Skips forward to move past the likely end of a declaration or statement. // // Looks forward, skipping over any matched symbol groups, to find the next - // position that is likely past the end of a declaration. This is a heuristic - // and should only be called when skipping past parse errors. + // position that is likely past the end of a declaration or statement. This + // is a heuristic and should only be called when skipping past parse errors. // // The strategy for recognizing when we have likely passed the end of a - // declaration: - // - If we get to close curly brace, we likely ended the entire context of - // declarations. - // - If we get to a semicolon, that should have ended the declaration. + // declaration or statement: + // - If we get to close curly brace, we likely ended the entire context. + // - If we get to a semicolon, that should have ended the declaration or + // statement. // - If we get to a new line from the `SkipRoot` token, but with the same or // less indentation, there is likely a missing semicolon. Continued - // declarations across multiple lines should be indented. + // declarations or statements across multiple lines should be indented. // - // If we find a semicolon based on this skipping, we try to build a parse node - // to represent it and will return that node. Otherwise we will return an - // empty optional. If `IsInsideDeclaration` is true (the default) we build a - // node that marks the end of the declaration we are inside. Otherwise we - // build an empty declaration node. - auto SkipPastLikelyDeclarationEnd(TokenizedBuffer::Token skip_root, - bool is_inside_declaration = true) + // If we find a semicolon based on this skipping, we call `on_semi_` to try + // to build a parse node to represent it, and will return that node. + // Otherwise we will return an empty optional. + auto SkipPastLikelyEnd(TokenizedBuffer::Token skip_root, SemiHandler on_semi) -> llvm::Optional; // Parses the signature of the function, consisting of a parameter list and an @@ -126,6 +130,36 @@ class ParseTree::Parser { // even when a node is returned. auto ParseDeclaration() -> llvm::Optional; + // Parses a parenthesized expression. + auto ParseParenExpression() -> llvm::Optional; + + // Parses a primary expression, which is either a terminal portion of an + // expression tree, such as an identifier or literal, or a parenthesized + // expression. + auto ParsePrimaryExpression() -> llvm::Optional; + + // Parses a designator expression suffix starting with `.`. + auto ParseDesignatorExpression(SubtreeStart start, bool has_errors) + -> llvm::Optional; + + // Parses a call expression suffix starting with `(`. + auto ParseCallExpression(SubtreeStart start, bool has_errors) + -> llvm::Optional; + + // Parses a postfix expression, which is a primary expression followed by + // zero or more of the following: + // + // - function applications + // - array indexes (TODO) + // - designators + auto ParsePostfixExpression() -> llvm::Optional; + + // Parses an expression. + auto ParseExpression() -> llvm::Optional; + + // Parses an expression statement: an expression followed by a semicolon. + auto ParseExpressionStatement() -> llvm::Optional; + ParseTree& tree; TokenizedBuffer& tokens; TokenDiagnosticEmitter& emitter;