From a83c22288f72a63b769df949609400421bdbe288 Mon Sep 17 00:00:00 2001 From: Richard Smith Date: Mon, 2 Aug 2021 15:37:43 -0700 Subject: [PATCH] [toolchain] Implement lexing and parsing support for #543. (#693) Lex [iuf][1-9][0-9]* as a new kind of "sized type literal" token. When parsing that token, form a literal expression. Co-authored-by: Chandler Carruth --- toolchain/lexer/token_kind.cpp | 6 + toolchain/lexer/token_kind.h | 3 + toolchain/lexer/token_registry.def | 3 + toolchain/lexer/tokenized_buffer.cpp | 66 +++++++++++ toolchain/lexer/tokenized_buffer.h | 7 +- toolchain/lexer/tokenized_buffer_test.cpp | 107 ++++++++++++++++++ toolchain/parser/parse_tree_test.cpp | 127 +++++++++++----------- toolchain/parser/parser_impl.cpp | 3 + 8 files changed, 257 insertions(+), 65 deletions(-) diff --git a/toolchain/lexer/token_kind.cpp b/toolchain/lexer/token_kind.cpp index 7ce6234e211a..6928e7a9b6c4 100644 --- a/toolchain/lexer/token_kind.cpp +++ b/toolchain/lexer/token_kind.cpp @@ -90,6 +90,12 @@ auto TokenKind::IsKeyword() const -> bool { return Table[static_cast(kind_value)]; } +auto TokenKind::IsSizedTypeLiteral() const -> bool { + return *this == TokenKind::IntegerTypeLiteral() || + *this == TokenKind::UnsignedIntegerTypeLiteral() || + *this == TokenKind::FloatingPointTypeLiteral(); +} + auto TokenKind::GetFixedSpelling() const -> llvm::StringRef { static constexpr llvm::StringLiteral Table[] = { #define CARBON_TOKEN(TokenName) "", diff --git a/toolchain/lexer/token_kind.h b/toolchain/lexer/token_kind.h index 002a03cb7878..b3679e0aca82 100644 --- a/toolchain/lexer/token_kind.h +++ b/toolchain/lexer/token_kind.h @@ -75,6 +75,9 @@ class TokenKind { // Test whether this kind of token is a keyword. [[nodiscard]] auto IsKeyword() const -> bool; + // Test whether this kind of token is a sized type literal. + [[nodiscard]] auto IsSizedTypeLiteral() const -> bool; + // If this token kind has a fixed spelling when in source code, returns it. // Otherwise returns an empty string. [[nodiscard]] auto GetFixedSpelling() const -> llvm::StringRef; diff --git a/toolchain/lexer/token_registry.def b/toolchain/lexer/token_registry.def index 7814d2b8e23a..ebbfb748816d 100644 --- a/toolchain/lexer/token_registry.def +++ b/toolchain/lexer/token_registry.def @@ -159,6 +159,9 @@ CARBON_TOKEN(Identifier) CARBON_TOKEN(IntegerLiteral) CARBON_TOKEN(RealLiteral) CARBON_TOKEN(StringLiteral) +CARBON_TOKEN(IntegerTypeLiteral) +CARBON_TOKEN(UnsignedIntegerTypeLiteral) +CARBON_TOKEN(FloatingPointTypeLiteral) CARBON_TOKEN(Error) CARBON_TOKEN(EndOfFile) diff --git a/toolchain/lexer/tokenized_buffer.cpp b/toolchain/lexer/tokenized_buffer.cpp index 282569d42760..08db26a50b71 100644 --- a/toolchain/lexer/tokenized_buffer.cpp +++ b/toolchain/lexer/tokenized_buffer.cpp @@ -369,6 +369,48 @@ class TokenizedBuffer::Lexer { return token; } + // Given a word that has already been lexed, determine whether it is a type + // literal and if so form the corresponding token. + auto LexWordAsTypeLiteralToken(llvm::StringRef word, int column) + -> LexResult { + if (word.size() < 2) { + // Too short to form one of these tokens. + return LexResult::NoMatch(); + } + if (!('1' <= word[1] && word[1] <= '9')) { + // Doesn't start with a valid initial digit. + return LexResult::NoMatch(); + } + + llvm::Optional kind; + switch (word.front()) { + case 'i': + kind = TokenKind::IntegerTypeLiteral(); + break; + case 'u': + kind = TokenKind::UnsignedIntegerTypeLiteral(); + break; + case 'f': + kind = TokenKind::FloatingPointTypeLiteral(); + break; + default: + return LexResult::NoMatch(); + }; + + llvm::StringRef suffix = word.substr(1); + llvm::APInt suffix_value; + if (suffix.getAsInteger(10, suffix_value)) { + return LexResult::NoMatch(); + } + + auto token = buffer.AddToken( + {.kind = *kind, .token_line = current_line, .column = column}); + buffer.GetTokenInfo(token).literal_index = + buffer.literal_int_storage.size(); + buffer.literal_int_storage.push_back(std::move(suffix_value)); + return token; + } + // Closes all open groups that cannot remain open across the symbol `K`. // Users may pass `Error` to close all open groups. auto CloseInvalidOpenGroups(TokenKind kind) -> void { @@ -431,6 +473,12 @@ class TokenizedBuffer::Lexer { current_column += identifier_text.size(); source_text = source_text.drop_front(identifier_text.size()); + // Check if the text is a type literal, and if so form such a literal. + if (LexResult result = + LexWordAsTypeLiteralToken(identifier_text, identifier_column)) { + return result; + } + // Check if the text matches a keyword token, and if so use that. TokenKind kind = llvm::StringSwitch(identifier_text) #define CARBON_KEYWORD_TOKEN(Name, Spelling) .Case(Spelling, TokenKind::Name()) @@ -583,6 +631,16 @@ auto TokenizedBuffer::GetTokenText(Token token) const -> llvm::StringRef { return relexed_token->Text(); } + // Refer back to the source text to avoid needing to reconstruct the + // spelling from the size. + if (token_info.kind.IsSizedTypeLiteral()) { + auto& line_info = GetLineInfo(token_info.token_line); + int64_t token_start = line_info.start + token_info.column; + llvm::StringRef suffix = + source->Text().substr(token_start + 1).take_while(IsDecimalDigit); + return llvm::StringRef(suffix.data() - 1, suffix.size() + 1); + } + if (token_info.kind == TokenKind::EndOfFile()) { return llvm::StringRef(); } @@ -630,6 +688,14 @@ auto TokenizedBuffer::GetStringLiteral(Token token) const -> llvm::StringRef { return literal_string_storage[token_info.literal_index]; } +auto TokenizedBuffer::GetTypeLiteralSize(Token token) const + -> const llvm::APInt& { + auto& token_info = GetTokenInfo(token); + assert(token_info.kind.IsSizedTypeLiteral() && + "The token must be a sized type literal!"); + return literal_int_storage[token_info.literal_index]; +} + auto TokenizedBuffer::GetMatchedClosingToken(Token opening_token) const -> Token { auto& opening_token_info = GetTokenInfo(opening_token); diff --git a/toolchain/lexer/tokenized_buffer.h b/toolchain/lexer/tokenized_buffer.h index feda2e9d357d..02aef7bdec48 100644 --- a/toolchain/lexer/tokenized_buffer.h +++ b/toolchain/lexer/tokenized_buffer.h @@ -293,6 +293,10 @@ class TokenizedBuffer { // Returns the value of a `StringLiteral()` token. [[nodiscard]] auto GetStringLiteral(Token token) const -> llvm::StringRef; + // Returns the size specified in a `*TypeLiteral()` token. + [[nodiscard]] auto GetTypeLiteralSize(Token token) const + -> const llvm::APInt&; + // Returns the closing token matched with the given opening token. // // The given token must be an opening token kind. @@ -453,7 +457,8 @@ class TokenizedBuffer { llvm::SmallVector identifier_infos; - // Storage for integers that form part of the value of a numeric literal. + // Storage for integers that form part of the value of a numeric or type + // literal. llvm::SmallVector literal_int_storage; llvm::SmallVector literal_string_storage; diff --git a/toolchain/lexer/tokenized_buffer_test.cpp b/toolchain/lexer/tokenized_buffer_test.cpp index 07297beda612..70aa242aca0d 100644 --- a/toolchain/lexer/tokenized_buffer_test.cpp +++ b/toolchain/lexer/tokenized_buffer_test.cpp @@ -833,6 +833,113 @@ TEST_F(LexerTest, InvalidStringLiterals) { } } +TEST_F(LexerTest, TypeLiterals) { + llvm::StringLiteral testcase = R"( + i0 i1 i20 i999999999999 i0x1 + u0 u1 u64 u64b + f32 f80 f1 fi + s1 + )"; + + auto buffer = Lex(testcase); + EXPECT_FALSE(buffer.HasErrors()); + ASSERT_THAT(buffer, + HasTokens(llvm::ArrayRef{ + {.kind = TokenKind::Identifier(), + .line = 2, + .column = 5, + .indent_column = 5, + .text = {"i0"}}, + {.kind = TokenKind::IntegerTypeLiteral(), + .line = 2, + .column = 8, + .indent_column = 5, + .text = {"i1"}}, + {.kind = TokenKind::IntegerTypeLiteral(), + .line = 2, + .column = 11, + .indent_column = 5, + .text = {"i20"}}, + {.kind = TokenKind::IntegerTypeLiteral(), + .line = 2, + .column = 15, + .indent_column = 5, + .text = {"i999999999999"}}, + {.kind = TokenKind::Identifier(), + .line = 2, + .column = 29, + .indent_column = 5, + .text = {"i0x1"}}, + + {.kind = TokenKind::Identifier(), + .line = 3, + .column = 5, + .indent_column = 5, + .text = {"u0"}}, + {.kind = TokenKind::UnsignedIntegerTypeLiteral(), + .line = 3, + .column = 8, + .indent_column = 5, + .text = {"u1"}}, + {.kind = TokenKind::UnsignedIntegerTypeLiteral(), + .line = 3, + .column = 11, + .indent_column = 5, + .text = {"u64"}}, + {.kind = TokenKind::Identifier(), + .line = 3, + .column = 15, + .indent_column = 5, + .text = {"u64b"}}, + + {.kind = TokenKind::FloatingPointTypeLiteral(), + .line = 4, + .column = 5, + .indent_column = 5, + .text = {"f32"}}, + {.kind = TokenKind::FloatingPointTypeLiteral(), + .line = 4, + .column = 9, + .indent_column = 5, + .text = {"f80"}}, + {.kind = TokenKind::FloatingPointTypeLiteral(), + .line = 4, + .column = 13, + .indent_column = 5, + .text = {"f1"}}, + {.kind = TokenKind::Identifier(), + .line = 4, + .column = 16, + .indent_column = 5, + .text = {"fi"}}, + + {.kind = TokenKind::Identifier(), + .line = 5, + .column = 5, + .indent_column = 5, + .text = {"s1"}}, + + {.kind = TokenKind::EndOfFile(), .line = 6, .column = 3}, + })); + + auto token_i1 = buffer.Tokens().begin() + 1; + EXPECT_EQ(buffer.GetTypeLiteralSize(*token_i1), 1); + auto token_i20 = buffer.Tokens().begin() + 2; + EXPECT_EQ(buffer.GetTypeLiteralSize(*token_i20), 20); + auto token_i999999999999 = buffer.Tokens().begin() + 3; + EXPECT_EQ(buffer.GetTypeLiteralSize(*token_i999999999999), 999999999999ull); + auto token_u1 = buffer.Tokens().begin() + 6; + EXPECT_EQ(buffer.GetTypeLiteralSize(*token_u1), 1); + auto token_u64 = buffer.Tokens().begin() + 7; + EXPECT_EQ(buffer.GetTypeLiteralSize(*token_u64), 64); + auto token_f32 = buffer.Tokens().begin() + 9; + EXPECT_EQ(buffer.GetTypeLiteralSize(*token_f32), 32); + auto token_f80 = buffer.Tokens().begin() + 10; + EXPECT_EQ(buffer.GetTypeLiteralSize(*token_f80), 80); + auto token_f1 = buffer.Tokens().begin() + 11; + EXPECT_EQ(buffer.GetTypeLiteralSize(*token_f1), 1); +} + TEST_F(LexerTest, Diagnostics) { llvm::StringLiteral testcase = R"( // Hello! diff --git a/toolchain/parser/parse_tree_test.cpp b/toolchain/parser/parse_tree_test.cpp index 0ebeb053ea6f..bde7fb8bbb6f 100644 --- a/toolchain/parser/parse_tree_test.cpp +++ b/toolchain/parser/parse_tree_test.cpp @@ -151,27 +151,27 @@ TEST_F(ParseTreeTest, } TEST_F(ParseTreeTest, FunctionDeclarationWithParameterList) { - TokenizedBuffer tokens = GetTokenizedBuffer("fn foo(bar: Int, baz: Int);"); + TokenizedBuffer tokens = GetTokenizedBuffer("fn foo(bar: i32, baz: i32);"); ParseTree tree = ParseTree::Parse(tokens, consumer); EXPECT_FALSE(tree.HasErrors()); - EXPECT_THAT(tree, - MatchParseTreeNodes( - {MatchFunctionDeclaration( - MatchDeclaredName("foo"), - MatchParameterList( - MatchPatternBinding(MatchDeclaredName("bar"), ":", - MatchNameReference("Int")), - MatchParameterListComma(), - MatchPatternBinding(MatchDeclaredName("baz"), ":", - MatchNameReference("Int")), - MatchParameterListEnd()), - MatchDeclarationEnd()), - MatchFileEnd()})); + EXPECT_THAT( + tree, + MatchParseTreeNodes( + {MatchFunctionDeclaration( + MatchDeclaredName("foo"), + MatchParameterList(MatchPatternBinding(MatchDeclaredName("bar"), + ":", MatchLiteral("i32")), + MatchParameterListComma(), + MatchPatternBinding(MatchDeclaredName("baz"), + ":", MatchLiteral("i32")), + MatchParameterListEnd()), + MatchDeclarationEnd()), + MatchFileEnd()})); } TEST_F(ParseTreeTest, FunctionDefinitionWithParameterList) { TokenizedBuffer tokens = GetTokenizedBuffer( - "fn foo(bar: Int, baz: Int) {\n" + "fn foo(bar: i64, baz: i64) {\n" " foo(baz, bar + baz);\n" "}"); ParseTree tree = ParseTree::Parse(tokens, consumer); @@ -181,13 +181,12 @@ TEST_F(ParseTreeTest, FunctionDefinitionWithParameterList) { MatchParseTreeNodes( {MatchFunctionDeclaration( MatchDeclaredName("foo"), - MatchParameterList( - MatchPatternBinding(MatchDeclaredName("bar"), ":", - MatchNameReference("Int")), - MatchParameterListComma(), - MatchPatternBinding(MatchDeclaredName("baz"), ":", - MatchNameReference("Int")), - MatchParameterListEnd()), + MatchParameterList(MatchPatternBinding(MatchDeclaredName("bar"), + ":", MatchLiteral("i64")), + MatchParameterListComma(), + MatchPatternBinding(MatchDeclaredName("baz"), + ":", MatchLiteral("i64")), + MatchParameterListEnd()), MatchCodeBlock( MatchExpressionStatement(MatchCallExpression( MatchNameReference("foo"), MatchNameReference("baz"), @@ -200,21 +199,21 @@ TEST_F(ParseTreeTest, FunctionDefinitionWithParameterList) { } TEST_F(ParseTreeTest, FunctionDeclarationWithReturnType) { - TokenizedBuffer tokens = GetTokenizedBuffer("fn foo() -> Int;"); + TokenizedBuffer tokens = GetTokenizedBuffer("fn foo() -> u32;"); ParseTree tree = ParseTree::Parse(tokens, consumer); EXPECT_FALSE(tree.HasErrors()); EXPECT_THAT( tree, MatchParseTreeNodes( {MatchFunctionDeclaration(MatchDeclaredName("foo"), MatchParameters(), - MatchReturnType(MatchNameReference("Int")), + MatchReturnType(MatchLiteral("u32")), MatchDeclarationEnd()), MatchFileEnd()})); } TEST_F(ParseTreeTest, FunctionDefinitionWithReturnType) { TokenizedBuffer tokens = GetTokenizedBuffer( - "fn foo() -> Int {\n" + "fn foo() -> f64 {\n" " return 42;\n" "}"); ParseTree tree = ParseTree::Parse(tokens, consumer); @@ -223,7 +222,7 @@ TEST_F(ParseTreeTest, FunctionDefinitionWithReturnType) { MatchParseTreeNodes( {MatchFunctionDeclaration( MatchDeclaredName("foo"), MatchParameters(), - MatchReturnType(MatchNameReference("Int")), + MatchReturnType(MatchLiteral("f64")), MatchCodeBlock(MatchReturnStatement(MatchLiteral("42"), MatchStatementEnd()), MatchCodeBlockEnd())), @@ -490,14 +489,14 @@ TEST_F(ParseTreeTest, Operators) { TEST_F(ParseTreeTest, OperatorFixity) { TokenizedBuffer tokens = GetTokenizedBuffer( - "fn F(p: Int*, n: Int) {\n" - " var q: Int* = p;\n" - " var t: Type = Int*;\n" + "fn F(p: i32*, n: i32) {\n" + " var q: i32* = p;\n" + " var t: Type = i32*;\n" " t = t**;\n" " n = n * n;\n" " n = n * *p;\n" " n = n*n;\n" - " G(Int*, n * n);\n" + " G(i32*, n * n);\n" "}"); ParseTree tree = ParseTree::Parse(tokens, consumer); EXPECT_FALSE(tree.HasErrors()); @@ -510,22 +509,22 @@ TEST_F(ParseTreeTest, OperatorFixity) { MatchParameters( MatchPatternBinding( MatchDeclaredName("p"), - MatchPostfixOperator(MatchNameReference("Int"), "*")), + MatchPostfixOperator(MatchLiteral("i32"), "*")), MatchParameterListComma(), MatchPatternBinding(MatchDeclaredName("n"), - MatchNameReference("Int"))), + MatchLiteral("i32"))), MatchCodeBlock( MatchVariableDeclaration( - MatchPatternBinding(MatchDeclaredName("q"), - MatchPostfixOperator( - MatchNameReference("Int"), "*")), + MatchPatternBinding( + MatchDeclaredName("q"), + MatchPostfixOperator(MatchLiteral("i32"), "*")), MatchVariableInitializer(MatchNameReference("p")), MatchDeclarationEnd()), MatchVariableDeclaration( MatchPatternBinding(MatchDeclaredName("t"), MatchNameReference("Type")), - MatchVariableInitializer(MatchPostfixOperator( - MatchNameReference("Int"), "*")), + MatchVariableInitializer( + MatchPostfixOperator(MatchLiteral("i32"), "*")), MatchDeclarationEnd()), MatchExpressionStatement(MatchInfixOperator( MatchNameReference("t"), "=", @@ -547,7 +546,7 @@ TEST_F(ParseTreeTest, OperatorFixity) { MatchNameReference("n")))), MatchExpressionStatement(MatchCallExpression( MatchNameReference("G"), - MatchPostfixOperator(MatchNameReference("Int"), "*"), + MatchPostfixOperator(MatchLiteral("i32"), "*"), MatchCallExpressionComma(), MatchInfixOperator(MatchNameReference("n"), "*", MatchNameReference("n")), @@ -565,31 +564,31 @@ TEST_F(ParseTreeTest, OperatorWhitespaceErrors) { const char* input; Kind kind; } testcases[] = { - {"var v: Type = Int*;", Valid}, - {"var v: Type = Int *;", Recovered}, - {"var v: Type = Int* ;", Valid}, - {"var v: Type = Int * ;", Recovered}, - {"var n: Int = n * n;", Valid}, - {"var n: Int = n*n;", Valid}, - {"var n: Int = (n)*3;", Valid}, - {"var n: Int = 3*(n);", Valid}, - {"var n: Int = n *n;", Recovered}, + {"var v: Type = i8*;", Valid}, + {"var v: Type = i8 *;", Recovered}, + {"var v: Type = i8* ;", Valid}, + {"var v: Type = i8 * ;", Recovered}, + {"var n: i8 = n * n;", Valid}, + {"var n: i8 = n*n;", Valid}, + {"var n: i8 = (n)*3;", Valid}, + {"var n: i8 = 3*(n);", Valid}, + {"var n: i8 = n *n;", Recovered}, // FIXME: We could figure out that this first Failed example is infix // with one-token lookahead. - {"var n: Int = n* n;", Failed}, - {"var n: Int = n* -n;", Failed}, - {"var n: Int = n* *p;", Failed}, + {"var n: i8 = n* n;", Failed}, + {"var n: i8 = n* -n;", Failed}, + {"var n: i8 = n* *p;", Failed}, // FIXME: We try to form (n*)*p and reject due to missing parentheses // before we notice the missing whitespace around the second `*`. // It'd be better to (somehow) form n*(*p) and reject due to the missing // whitespace around the first `*`. - {"var n: Int = n**p;", Failed}, - {"var n: Int = -n;", Valid}, - {"var n: Int = - n;", Recovered}, - {"var n: Int =-n;", Valid}, - {"var n: Int =- n;", Recovered}, - {"var n: Int = F(Int *);", Recovered}, - {"var n: Int = F(Int *, 0);", Recovered}, + {"var n: i8 = n**p;", Failed}, + {"var n: i8 = -n;", Valid}, + {"var n: i8 = - n;", Recovered}, + {"var n: i8 =-n;", Valid}, + {"var n: i8 =- n;", Recovered}, + {"var n: i8 = F(i8 *);", Recovered}, + {"var n: i8 = F(i8 *, 0);", Recovered}, }; for (auto [input, kind] : testcases) { @@ -603,8 +602,8 @@ TEST_F(ParseTreeTest, OperatorWhitespaceErrors) { TEST_F(ParseTreeTest, VariableDeclarations) { TokenizedBuffer tokens = GetTokenizedBuffer( - "var v: Int = 0;\n" - "var w: Int;\n" + "var v: i32 = 0;\n" + "var w: i32;\n" "fn F() {\n" " var s: String = \"hello\";\n" "}"); @@ -615,12 +614,12 @@ TEST_F(ParseTreeTest, VariableDeclarations) { MatchParseTreeNodes( {MatchVariableDeclaration( MatchPatternBinding(MatchDeclaredName("v"), ":", - MatchNameReference("Int")), + MatchLiteral("i32")), MatchVariableInitializer(MatchLiteral("0")), MatchDeclarationEnd()), MatchVariableDeclaration( MatchPatternBinding(MatchDeclaredName("w"), ":", - MatchNameReference("Int")), + MatchLiteral("i32")), MatchDeclarationEnd()), MatchFunctionWithBody(MatchVariableDeclaration( MatchPatternBinding(MatchDeclaredName("s"), ":", @@ -914,7 +913,7 @@ TEST_F(ParseTreeTest, Return) { " return;\n" " }\n" "}\n" - "fn G(x: Int) -> Int {\n" + "fn G(x: Foo) -> Foo {\n" " return x;\n" "}"); ParseTree tree = ParseTree::Parse(tokens, consumer); @@ -930,8 +929,8 @@ TEST_F(ParseTreeTest, Return) { MatchFunctionDeclaration( MatchDeclaredName(), MatchParameters(MatchPatternBinding(MatchDeclaredName("x"), ":", - MatchNameReference("Int"))), - MatchReturnType(MatchNameReference("Int")), + MatchNameReference("Foo"))), + MatchReturnType(MatchNameReference("Foo")), MatchCodeBlock(MatchReturnStatement(MatchNameReference("x"), MatchStatementEnd()), MatchCodeBlockEnd())), diff --git a/toolchain/parser/parser_impl.cpp b/toolchain/parser/parser_impl.cpp index 320bee94d51e..27bec54f66c1 100644 --- a/toolchain/parser/parser_impl.cpp +++ b/toolchain/parser/parser_impl.cpp @@ -678,6 +678,9 @@ auto ParseTree::Parser::ParsePrimaryExpression() -> llvm::Optional { case TokenKind::IntegerLiteral(): case TokenKind::RealLiteral(): case TokenKind::StringLiteral(): + case TokenKind::IntegerTypeLiteral(): + case TokenKind::UnsignedIntegerTypeLiteral(): + case TokenKind::FloatingPointTypeLiteral(): kind = ParseNodeKind::Literal(); break;