Initial support for CR+LF (DOS / Windows) line endings. (#4056)

It turns out we can make these work with very minimal complexity because
the LF is still in the right place either way. This also lets us easily
support mixtures of LF and CR+LF line endings gracefully. We create the
line structures around the LF bytes and then have the byte-dispatch loop
notice a CR followed by an LF and skip to the LF behavior.

Rather than add the remaining complexity around supporting bare CR and
LF+CR sequences (both of which are quite rare now), this just adds
diagnostics when we encounter a CR byte that won't fall out of our CR+LF
handling. This is a better experience for users than the alternative. We
still have a TODO to handle the full complexity of vertical whitespace,
but I've updated it to reflect that the common case should be handled
already.

This isn't complete though: we need to add support in string literal
lexing, and we need to teach the diagnostic rendering to handle the
error messages above better. But those will be future PRs, this is
enough to unblock folks who happen to edit a Carbon source file with
notepad on Windows which seems important.

---------

Co-authored-by: Richard Smith <richard@metafoo.co.uk>
This commit is contained in:
Chandler Carruth
2024-06-18 02:23:25 +00:00
committed by GitHub
co-authored by Richard Smith
parent 1133f3b8b7
commit 565fc5cebb
3 changed files with 128 additions and 5 deletions
+82
View File
@@ -101,6 +101,88 @@ TEST_F(LexerTest, TracksLinesAndColumns) {
}));
}
TEST_F(LexerTest, TracksLinesAndColumnsCRLF) {
auto buffer =
Lex("\r\n ;;\r\n ;;;\r\n x\"foo\" '''baz\r\n a\r\n ''' y");
EXPECT_FALSE(buffer.has_errors());
EXPECT_THAT(
buffer,
HasTokens(llvm::ArrayRef<ExpectedToken>{
{.kind = TokenKind::FileStart,
.line = 1,
.column = 1,
.indent_column = 1},
{.kind = TokenKind::Semi, .line = 2, .column = 3, .indent_column = 3},
{.kind = TokenKind::Semi, .line = 2, .column = 4, .indent_column = 3},
{.kind = TokenKind::Semi, .line = 3, .column = 4, .indent_column = 4},
{.kind = TokenKind::Semi, .line = 3, .column = 5, .indent_column = 4},
{.kind = TokenKind::Semi, .line = 3, .column = 6, .indent_column = 4},
{.kind = TokenKind::Identifier,
.line = 4,
.column = 4,
.indent_column = 4,
.text = "x"},
{.kind = TokenKind::StringLiteral,
.line = 4,
.column = 5,
.indent_column = 4},
{.kind = TokenKind::StringLiteral,
.line = 4,
.column = 11,
.indent_column = 4},
{.kind = TokenKind::Identifier,
.line = 6,
.column = 6,
.indent_column = 11,
.text = "y"},
{.kind = TokenKind::FileEnd, .line = 6, .column = 7},
}));
}
TEST_F(LexerTest, InvalidCR) {
auto buffer = Lex("\n ;;\r ;\n x");
EXPECT_TRUE(buffer.has_errors());
EXPECT_THAT(
buffer,
HasTokens(llvm::ArrayRef<ExpectedToken>{
{.kind = TokenKind::FileStart,
.line = 1,
.column = 1,
.indent_column = 1},
{.kind = TokenKind::Semi, .line = 2, .column = 2, .indent_column = 2},
{.kind = TokenKind::Semi, .line = 2, .column = 3, .indent_column = 2},
{.kind = TokenKind::Semi, .line = 2, .column = 6, .indent_column = 2},
{.kind = TokenKind::Identifier,
.line = 3,
.column = 4,
.indent_column = 4,
.text = "x"},
{.kind = TokenKind::FileEnd, .line = 3, .column = 5},
}));
}
TEST_F(LexerTest, InvalidLFCR) {
auto buffer = Lex("\n ;;\n\r ;\n x");
EXPECT_TRUE(buffer.has_errors());
EXPECT_THAT(
buffer,
HasTokens(llvm::ArrayRef<ExpectedToken>{
{.kind = TokenKind::FileStart,
.line = 1,
.column = 1,
.indent_column = 1},
{.kind = TokenKind::Semi, .line = 2, .column = 2, .indent_column = 2},
{.kind = TokenKind::Semi, .line = 2, .column = 3, .indent_column = 2},
{.kind = TokenKind::Semi, .line = 3, .column = 3, .indent_column = 1},
{.kind = TokenKind::Identifier,
.line = 4,
.column = 4,
.indent_column = 4,
.text = "x"},
{.kind = TokenKind::FileEnd, .line = 4, .column = 5},
}));
}
TEST_F(LexerTest, HandlesNumericLiteral) {
auto buffer = Lex("12-578\n 1 2\n0x12_3ABC\n0b10_10_11\n1_234_567\n1.5e9");
EXPECT_FALSE(buffer.has_errors());