mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-10-05 06:11:05 +01:00
The presence or absence of whitespace is used to determine which operator is in use, following the rules described in #520. Support for prefix * dereference operator follows #523. Co-authored-by: Geoff Romer <gromer@google.com>
This commit is contained in:
committed by
GitHub
co-authored by
Geoff Romer
parent
d6b47ba10f
commit
89e21113c3
@@ -18,6 +18,13 @@ SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
*/
|
||||
%option yylineno
|
||||
|
||||
/* Lexing a token immediately after consuming some whitespace. */
|
||||
%s AFTER_WHITESPACE
|
||||
/* Lexing a token immediately after consuming an operand-ending token:
|
||||
* a closing bracket, identifier, or literal.
|
||||
*/
|
||||
%s AFTER_OPERAND
|
||||
|
||||
AND "and"
|
||||
ARROW "->"
|
||||
AUTO "auto"
|
||||
@@ -53,13 +60,21 @@ AWAIT "__await"
|
||||
identifier [A-Za-z_][A-Za-z0-9_]*
|
||||
integer_literal [0-9]+
|
||||
horizontal_whitespace [ \t\r]
|
||||
whitespace [ \t\r\n]
|
||||
operand_start [(A-Za-z0-9_"]
|
||||
|
||||
%{
|
||||
// This macro is expanded to run each time a token is recognized.
|
||||
// This macro is expanded immediately before each action specified below.
|
||||
//
|
||||
// Advances the current token position by yyleng columns without changing
|
||||
// the line number.
|
||||
# define YY_USER_ACTION context.current_token_position.columns(yyleng);
|
||||
// the line number, and takes us out of the after-whitespace / after-operand
|
||||
// state.
|
||||
# define YY_USER_ACTION \
|
||||
context.current_token_position.columns(yyleng); \
|
||||
if (YY_START == AFTER_WHITESPACE || \
|
||||
YY_START == AFTER_OPERAND) { \
|
||||
BEGIN(INITIAL); \
|
||||
}
|
||||
%}
|
||||
|
||||
%%
|
||||
@@ -105,20 +120,58 @@ horizontal_whitespace [ \t\r]
|
||||
"=" return yy::parser::make_EQUAL(context.current_token_position);
|
||||
"-" return yy::parser::make_MINUS(context.current_token_position);
|
||||
"+" return yy::parser::make_PLUS(context.current_token_position);
|
||||
"*" return yy::parser::make_STAR(context.current_token_position);
|
||||
"/" return yy::parser::make_SLASH(context.current_token_position);
|
||||
"(" return yy::parser::make_LEFT_PARENTHESIS(context.current_token_position);
|
||||
")" return yy::parser::make_RIGHT_PARENTHESIS(context.current_token_position);
|
||||
")" { BEGIN(AFTER_OPERAND); return yy::parser::make_RIGHT_PARENTHESIS(context.current_token_position); }
|
||||
"{" return yy::parser::make_LEFT_CURLY_BRACE(context.current_token_position);
|
||||
"}" return yy::parser::make_RIGHT_CURLY_BRACE(context.current_token_position);
|
||||
"}" { BEGIN(AFTER_OPERAND); return yy::parser::make_RIGHT_CURLY_BRACE(context.current_token_position); }
|
||||
"[" return yy::parser::make_LEFT_SQUARE_BRACKET(context.current_token_position);
|
||||
"]" return yy::parser::make_RIGHT_SQUARE_BRACKET(context.current_token_position);
|
||||
"]" { BEGIN(AFTER_OPERAND); return yy::parser::make_RIGHT_SQUARE_BRACKET(context.current_token_position); }
|
||||
"." return yy::parser::make_PERIOD(context.current_token_position);
|
||||
"," return yy::parser::make_COMMA(context.current_token_position);
|
||||
";" return yy::parser::make_SEMICOLON(context.current_token_position);
|
||||
":" return yy::parser::make_COLON(context.current_token_position);
|
||||
|
||||
/*
|
||||
For a `*` operator, we look at whitespace and local context to determine the
|
||||
arity and fixity. There are two ways to write a binary operator:
|
||||
|
||||
1) Whitespace on both sides.
|
||||
2) Whitespace on neither side, and the previous token is considered to be
|
||||
the end of an operand, and the next token is considered to be the start
|
||||
of an operand.
|
||||
|
||||
Otherwise, the operator is unary, but we also check for whitespace to help
|
||||
the parser enforce the rule that whitespace is not permitted between the
|
||||
operator and its operand, leading to three more cases:
|
||||
|
||||
3) Whitespace before (but implicitly not after, because that would give a
|
||||
longer match and hit case 1): this can only be a prefix operator.
|
||||
4) Whitespace after and not before: this can only be a postfix operator.
|
||||
5) No whitespace on either side (otherwise the longest match would take us
|
||||
to case 4): this is a unary operator and could be either prefix or
|
||||
postfix.
|
||||
*/
|
||||
<AFTER_WHITESPACE>"*"{whitespace}+ /*case 1*/ {
|
||||
BEGIN(AFTER_WHITESPACE);
|
||||
return yy::parser::make_BINARY_STAR(context.current_token_position);
|
||||
}
|
||||
<AFTER_OPERAND>"*"/{operand_start} /*case 2*/ {
|
||||
return yy::parser::make_BINARY_STAR(context.current_token_position);
|
||||
}
|
||||
<AFTER_WHITESPACE>"*" /*case 3*/ {
|
||||
return yy::parser::make_PREFIX_STAR(context.current_token_position);
|
||||
}
|
||||
<INITIAL,AFTER_OPERAND>"*"{whitespace}+ /*case 4*/ {
|
||||
BEGIN(AFTER_WHITESPACE);
|
||||
return yy::parser::make_POSTFIX_STAR(context.current_token_position);
|
||||
}
|
||||
<INITIAL,AFTER_OPERAND>"*" /*case 5*/ {
|
||||
return yy::parser::make_UNARY_STAR(context.current_token_position);
|
||||
}
|
||||
|
||||
{identifier} {
|
||||
BEGIN(AFTER_OPERAND);
|
||||
int n = strlen(yytext);
|
||||
auto r = reinterpret_cast<char*>(malloc((n + 1) * sizeof(char)));
|
||||
strncpy(r, yytext, n + 1);
|
||||
@@ -126,6 +179,7 @@ horizontal_whitespace [ \t\r]
|
||||
}
|
||||
|
||||
{integer_literal} {
|
||||
BEGIN(AFTER_OPERAND);
|
||||
auto r = atof(yytext);
|
||||
return yy::parser::make_integer_literal(r, context.current_token_position);
|
||||
}
|
||||
@@ -140,6 +194,7 @@ horizontal_whitespace [ \t\r]
|
||||
{horizontal_whitespace}+ {
|
||||
// Make the span empty by setting start to end.
|
||||
context.current_token_position.step();
|
||||
BEGIN(AFTER_WHITESPACE);
|
||||
}
|
||||
|
||||
\n+ {
|
||||
@@ -147,6 +202,7 @@ horizontal_whitespace [ \t\r]
|
||||
context.current_token_position.lines(yyleng);
|
||||
// Make the span empty by setting start to end.
|
||||
context.current_token_position.step();
|
||||
BEGIN(AFTER_WHITESPACE);
|
||||
}
|
||||
|
||||
. {
|
||||
|
||||
@@ -123,7 +123,8 @@ void yy::parser::error(
|
||||
%token TYPE
|
||||
%token FN
|
||||
%token FNTY
|
||||
%token ARROW
|
||||
%token ARROW "->"
|
||||
%token FNARROW "-> in return type"
|
||||
%token VAR
|
||||
%token EQUAL_EQUAL
|
||||
%token IF
|
||||
@@ -142,14 +143,20 @@ void yy::parser::error(
|
||||
%token CHOICE
|
||||
%token MATCH
|
||||
%token CASE
|
||||
%token DBLARROW
|
||||
%token DBLARROW "=>"
|
||||
%token DEFAULT
|
||||
%token AUTO
|
||||
%token
|
||||
EQUAL "="
|
||||
MINUS "-"
|
||||
PLUS "+"
|
||||
STAR "*"
|
||||
// The lexer determines the arity and fixity of each `*` based on whitespace
|
||||
// and adjacent tokens. UNARY_STAR indicates that the operator is unary but
|
||||
// could be either prefix or postfix.
|
||||
UNARY_STAR "unary *"
|
||||
PREFIX_STAR "prefix *"
|
||||
POSTFIX_STAR "postfix *"
|
||||
BINARY_STAR "binary *"
|
||||
SLASH "/"
|
||||
LEFT_PARENTHESIS "("
|
||||
RIGHT_PARENTHESIS ")"
|
||||
@@ -163,12 +170,22 @@ void yy::parser::error(
|
||||
COLON ":"
|
||||
;
|
||||
|
||||
%precedence FNARROW
|
||||
%precedence "{" "}"
|
||||
%precedence ":" "," DBLARROW
|
||||
%left OR AND
|
||||
%nonassoc EQUAL_EQUAL
|
||||
%left "+" "-"
|
||||
%precedence NOT UNARY_MINUS
|
||||
%left BINARY_STAR
|
||||
%precedence NOT UNARY_MINUS PREFIX_STAR
|
||||
// We need to give the `UNARY_STAR` token a precedence, rather than overriding
|
||||
// the precedence of the `expression UNARY_STAR` rule below, because bison
|
||||
// compares the precedence of the final token (for a shift) to the precedence
|
||||
// of the other rule (for a reduce) when attempting to resolve a shift-reduce
|
||||
// conflict. See https://stackoverflow.com/a/26188429/1041090. When UNARY_STAR
|
||||
// is the final token of a rule, it must be a postfix usage, so we give it the
|
||||
// same precedence as POSTFIX_STAR.
|
||||
%precedence POSTFIX_STAR UNARY_STAR
|
||||
%left "." ARROW
|
||||
%precedence "(" ")" "[" "]"
|
||||
|
||||
@@ -214,6 +231,8 @@ expression:
|
||||
{ $$ = Carbon::Expression::MakeBinOp(yylineno, Carbon::Operator::Add, $1, $3); }
|
||||
| expression "-" expression
|
||||
{ $$ = Carbon::Expression::MakeBinOp(yylineno, Carbon::Operator::Sub, $1, $3); }
|
||||
| expression BINARY_STAR expression
|
||||
{ $$ = Carbon::Expression::MakeBinOp(yylineno, Carbon::Operator::Mul, $1, $3); }
|
||||
| expression AND expression
|
||||
{ $$ = Carbon::Expression::MakeBinOp(yylineno, Carbon::Operator::And, $1, $3); }
|
||||
| expression OR expression
|
||||
@@ -222,8 +241,16 @@ expression:
|
||||
{ $$ = Carbon::Expression::MakeUnOp(yylineno, Carbon::Operator::Not, $2); }
|
||||
| "-" expression %prec UNARY_MINUS
|
||||
{ $$ = Carbon::Expression::MakeUnOp(yylineno, Carbon::Operator::Neg, $2); }
|
||||
| PREFIX_STAR expression
|
||||
{ $$ = Carbon::Expression::MakeUnOp(yylineno, Carbon::Operator::Deref, $2); }
|
||||
| UNARY_STAR expression %prec PREFIX_STAR
|
||||
{ $$ = Carbon::Expression::MakeUnOp(yylineno, Carbon::Operator::Deref, $2); }
|
||||
| expression tuple
|
||||
{ $$ = Carbon::Expression::MakeCall(yylineno, $1, $2); }
|
||||
| expression POSTFIX_STAR
|
||||
{ $$ = Carbon::Expression::MakeUnOp(yylineno, Carbon::Operator::Ptr, $1); }
|
||||
| expression UNARY_STAR
|
||||
{ $$ = Carbon::Expression::MakeUnOp(yylineno, Carbon::Operator::Ptr, $1); }
|
||||
| FNTY tuple return_type
|
||||
{ $$ = Carbon::Expression::MakeFunType(yylineno, $2, $3); }
|
||||
;
|
||||
@@ -324,7 +351,7 @@ statement_list:
|
||||
return_type:
|
||||
// Empty
|
||||
{ $$ = Carbon::Expression::MakeUnit(yylineno); }
|
||||
| ARROW expression
|
||||
| ARROW expression %prec FNARROW
|
||||
{ $$ = $2; }
|
||||
;
|
||||
function_definition:
|
||||
|
||||
Reference in New Issue
Block a user