Add a start-of-file token and parse node. (#3263)

This removes a (very) hot branch in the lexer where we need to special
case when a token is the first token and can't look at its previous
token. It also seems like a generally nice change to the structure of
both the token buffer and parse tree as there are now bracketing
elements for both ends and we should be able to avoid similar branching
in the future.

Mostly mechanical updates to the lexer and parser code to handle this,
but also needed to special case the location information in the
autoupdate code. And then the usual large body of auto-updated tests.

No benchmark data for this change alone as in isolation and in the
current lexer structure it doesn't make a big difference. But this
branch was particularly difficult to handle when trying to update the
whitespace skipping code to be faster, and so I think it is worth
systematically avoiding the special case here.
This commit is contained in:
Chandler Carruth
2023-10-04 23:36:35 +00:00
committed by GitHub
parent 412e4fb461
commit a46ca6bf7a
208 changed files with 482 additions and 209 deletions
+16 -4
View File
@@ -268,9 +268,7 @@ class TokenizedBuffer::Lexer {
}
auto NoteWhitespace() -> void {
if (!buffer_->token_infos_.empty()) {
buffer_->token_infos_.back().has_trailing_space = true;
}
buffer_->token_infos_.back().has_trailing_space = true;
}
auto SkipWhitespace(llvm::StringRef& source_text) -> bool {
@@ -709,6 +707,15 @@ class TokenizedBuffer::Lexer {
.column = current_column_});
}
auto AddStartOfFileToken() -> void {
// Note that the start-of-file always has trailing space because it *is*
// whitespace.
buffer_->AddToken({.kind = TokenKind::StartOfFile,
.has_trailing_space = true,
.token_line = current_line_,
.column = current_column_});
}
constexpr static auto MakeDispatchTable() -> DispatchTableT {
DispatchTableT table = {};
auto dispatch_lex_error = +[](Lexer& lexer, llvm::StringRef& source_text) {
@@ -831,6 +838,10 @@ auto TokenizedBuffer::Lex(SourceBuffer& source, DiagnosticConsumer& consumer)
// dispatch structures.
constexpr Lexer::DispatchTableT DispatchTable = Lexer::MakeDispatchTable();
// Before lexing any source text, add the start-of-file token so that code can
// assume a non-empty token buffer for the rest of lexing.
lexer.AddStartOfFileToken();
llvm::StringRef source_text = source.text();
while (lexer.SkipWhitespace(source_text)) {
Lexer::LexResult result =
@@ -914,7 +925,8 @@ auto TokenizedBuffer::GetTokenText(Token token) const -> llvm::StringRef {
return llvm::StringRef(suffix.data() - 1, suffix.size() + 1);
}
if (token_info.kind == TokenKind::EndOfFile) {
if (token_info.kind == TokenKind::StartOfFile ||
token_info.kind == TokenKind::EndOfFile) {
return llvm::StringRef();
}