Clean up handling of incomplete line locations. (#3011)

Building on #3010, the handling of incomplete lines seems like it can be
straightened out. Doing this separately because it seemed better to
demonstrate tests aren't affected by the change.
This commit is contained in:
Jon Ross-Perkins
2023-07-24 23:17:58 +00:00
committed by GitHub
parent a3d52b089d
commit a7c2885728
5 changed files with 51 additions and 49 deletions
@@ -0,0 +1,22 @@
// Part of the Carbon Language project, under the Apache License v2.0 with LLVM
// Exceptions. See /LICENSE for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
// An autoupdate wants to put the CHECK in the middle of the block string.
// NOAUTOUPDATE
// CHECK:STDOUT: [
// CHECK:STDOUT: { index: 0, kind: 'Var', line: {{ *}}[[@LINE+12]], column: 1, indent: 1, spelling: 'var', has_trailing_space: true },
// CHECK:STDOUT: { index: 1, kind: 'Identifier', line: {{ *}}[[@LINE+11]], column: 5, indent: 1, spelling: 's', identifier: 0 },
// CHECK:STDOUT: { index: 2, kind: 'Colon', line: {{ *}}[[@LINE+10]], column: 6, indent: 1, spelling: ':', has_trailing_space: true },
// CHECK:STDOUT: { index: 3, kind: 'StringTypeLiteral', line: {{ *}}[[@LINE+9]], column: 8, indent: 1, spelling: 'String', has_trailing_space: true },
// CHECK:STDOUT: { index: 4, kind: 'Equal', line: {{ *}}[[@LINE+8]], column: 15, indent: 1, spelling: '=', has_trailing_space: true },
// CHECK:STDOUT: { index: 5, kind: 'StringLiteral', line: {{ *}}[[@LINE+7]], column: 17, indent: 1, spelling: ''''
// CHECK:STDOUT: error here: '''', value: `error here: `, has_trailing_space: true },
// CHECK:STDOUT: { index: 6, kind: 'EndOfFile', line: {{ *}}[[@LINE+6]], column: {{[0-9]+}}, indent: 17, spelling: '' },
// CHECK:STDOUT: ]
// CHECK:STDERR: fail_block_string_second_line.carbon:[[@LINE+4]]:3: Only whitespace is permitted before the closing `'''` of a multi-line string.
// CHECK:STDERR: error here: '''
// CHECK:STDERR: ^
var s: String = '''
error here: '''
+19 -32
View File
@@ -84,9 +84,9 @@ class TokenizedBuffer::Lexer {
Lexer(TokenizedBuffer& buffer, DiagnosticConsumer& consumer)
: buffer_(&buffer),
translator_(&buffer, &current_column_),
translator_(&buffer),
emitter_(translator_, consumer),
token_translator_(&buffer, &current_column_),
token_translator_(&buffer),
token_emitter_(token_translator_, consumer),
current_line_(buffer.AddLine(LineInfo(0))),
current_line_info_(&buffer.GetLineInfo(current_line_)) {}
@@ -897,8 +897,6 @@ auto TokenizedBuffer::SourceBufferLocationTranslator::GetLocation(
const auto* line_it = std::partition_point(
buffer_->line_infos_.begin(), buffer_->line_infos_.end(),
[offset](const LineInfo& line) { return line.start <= offset; });
bool incomplete_line_info = last_line_lexed_to_column_ != nullptr &&
line_it == buffer_->line_infos_.end();
// Step back one line to find the line containing the given position.
CARBON_CHECK(line_it != buffer_->line_infos_.begin())
@@ -907,33 +905,23 @@ auto TokenizedBuffer::SourceBufferLocationTranslator::GetLocation(
int line_number = line_it - buffer_->line_infos_.begin();
int column_number = offset - line_it->start;
llvm::StringRef line;
// We might still be lexing the last line. If so, check to see if there are
// any newline characters between the position we've finished lexing up to
// and the given location.
if (incomplete_line_info && column_number > *last_line_lexed_to_column_) {
column_number = *last_line_lexed_to_column_;
int64_t start = line_it->start;
for (int64_t i = line_it->start + *last_line_lexed_to_column_; i != offset;
++i) {
if (buffer_->source_->text()[i] == '\n') {
start = i;
++line_number;
column_number = 0;
} else {
++column_number;
}
// Start by grabbing the line from the buffer. If the line isn't fully lexed,
// the length will be npos and the line will be grabbed from the known start
// to the end of the buffer; we'll then adjust the length.
llvm::StringRef line =
buffer_->source_->text().substr(line_it->start, line_it->length);
if (line_it->length == static_cast<int32_t>(llvm::StringRef::npos)) {
CARBON_CHECK(line.take_front(column_number).count('\n') == 0)
<< "Currently we assume no unlexed newlines prior to the error column, "
"but there was one when erroring at "
<< buffer_->source_->filename() << ":" << line_number << ":"
<< column_number;
// Look for the next newline since we don't know the length. We can start at
// the column because prior newlines will have been lexed.
auto end_newline_pos = line.find('\n', column_number);
if (end_newline_pos != llvm::StringRef::npos) {
line = line.take_front(end_newline_pos);
}
line = buffer_->source_->text().substr(start).take_until(
[](char c) { return c == '\n'; });
} else if (line_it->length < 0) {
line =
buffer_->source_->text().substr(line_it->start).take_until([](char c) {
return c == '\n';
});
} else {
line = buffer_->source_->text().substr(line_it->start, line_it->length);
}
return {.file_name = buffer_->source_->filename(),
@@ -953,8 +941,7 @@ auto TokenizedBuffer::TokenLocationTranslator::GetLocation(Token token)
// Find the corresponding file location.
// TODO: Should we somehow indicate in the diagnostic location if this token
// is a recovery token that doesn't correspond to the original source?
return SourceBufferLocationTranslator(buffer_, last_line_lexed_to_column_)
.GetLocation(token_start);
return SourceBufferLocationTranslator(buffer_).GetLocation(token_start);
}
} // namespace Carbon
+8 -14
View File
@@ -170,18 +170,14 @@ class TokenizedBuffer {
// buffer locations.
class TokenLocationTranslator : public DiagnosticLocationTranslator<Token> {
public:
explicit TokenLocationTranslator(const TokenizedBuffer* buffer,
int* last_line_lexed_to_column)
: buffer_(buffer),
last_line_lexed_to_column_(last_line_lexed_to_column) {}
explicit TokenLocationTranslator(const TokenizedBuffer* buffer)
: buffer_(buffer) {}
// Map the given token into a diagnostic location.
auto GetLocation(Token token) -> DiagnosticLocation override;
private:
const TokenizedBuffer* buffer_;
// Passed to SourceBufferLocationTranslator.
int* last_line_lexed_to_column_;
};
// Lexes a buffer of source code into a tokenized buffer.
@@ -298,10 +294,8 @@ class TokenizedBuffer {
class SourceBufferLocationTranslator
: public DiagnosticLocationTranslator<const char*> {
public:
explicit SourceBufferLocationTranslator(const TokenizedBuffer* buffer,
int* last_line_lexed_to_column)
: buffer_(buffer),
last_line_lexed_to_column_(last_line_lexed_to_column) {}
explicit SourceBufferLocationTranslator(const TokenizedBuffer* buffer)
: buffer_(buffer) {}
// Map the given position within the source buffer into a diagnostic
// location.
@@ -309,9 +303,6 @@ class TokenizedBuffer {
private:
const TokenizedBuffer* buffer_;
// The last lexed column, for determining whether the last line should be
// checked for unlexed newlines. May be null after lexing is complete.
int* last_line_lexed_to_column_;
};
// Specifies minimum widths to use when printing a token's fields via
@@ -360,7 +351,10 @@ class TokenizedBuffer {
struct LineInfo {
// The length will always be assigned later. Indent may be assigned if
// non-zero.
explicit LineInfo(int64_t start) : start(start), length(-1), indent(0) {}
explicit LineInfo(int64_t start)
: start(start),
length(static_cast<int32_t>(llvm::StringRef::npos)),
indent(0) {}
// Zero-based byte offset of the start of the line within the source buffer
// provided.
+1 -2
View File
@@ -20,8 +20,7 @@ namespace Carbon {
auto ParseTree::Parse(TokenizedBuffer& tokens, DiagnosticConsumer& consumer,
llvm::raw_ostream* vlog_stream) -> ParseTree {
TokenizedBuffer::TokenLocationTranslator translator(
&tokens, /*last_line_lexed_to_column=*/nullptr);
TokenizedBuffer::TokenLocationTranslator translator(&tokens);
TokenDiagnosticEmitter emitter(translator, consumer);
// Delegate to the parser.
@@ -14,7 +14,7 @@ class ParseTreeNodeLocationTranslator
public:
explicit ParseTreeNodeLocationTranslator(const TokenizedBuffer* tokens,
const ParseTree* parse_tree)
: token_translator_(tokens, nullptr), parse_tree_(parse_tree) {}
: token_translator_(tokens), parse_tree_(parse_tree) {}
// Map the given token into a diagnostic location.
auto GetLocation(ParseTree::Node node) -> DiagnosticLocation override {