diff --git a/toolchain/lexer/testdata/fail_block_string_second_line.carbon b/toolchain/lexer/testdata/fail_block_string_second_line.carbon new file mode 100644 index 000000000000..63971acee8d5 --- /dev/null +++ b/toolchain/lexer/testdata/fail_block_string_second_line.carbon @@ -0,0 +1,22 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +// An autoupdate wants to put the CHECK in the middle of the block string. +// NOAUTOUPDATE + +// CHECK:STDOUT: [ +// CHECK:STDOUT: { index: 0, kind: 'Var', line: {{ *}}[[@LINE+12]], column: 1, indent: 1, spelling: 'var', has_trailing_space: true }, +// CHECK:STDOUT: { index: 1, kind: 'Identifier', line: {{ *}}[[@LINE+11]], column: 5, indent: 1, spelling: 's', identifier: 0 }, +// CHECK:STDOUT: { index: 2, kind: 'Colon', line: {{ *}}[[@LINE+10]], column: 6, indent: 1, spelling: ':', has_trailing_space: true }, +// CHECK:STDOUT: { index: 3, kind: 'StringTypeLiteral', line: {{ *}}[[@LINE+9]], column: 8, indent: 1, spelling: 'String', has_trailing_space: true }, +// CHECK:STDOUT: { index: 4, kind: 'Equal', line: {{ *}}[[@LINE+8]], column: 15, indent: 1, spelling: '=', has_trailing_space: true }, +// CHECK:STDOUT: { index: 5, kind: 'StringLiteral', line: {{ *}}[[@LINE+7]], column: 17, indent: 1, spelling: '''' +// CHECK:STDOUT: error here: '''', value: `error here: `, has_trailing_space: true }, +// CHECK:STDOUT: { index: 6, kind: 'EndOfFile', line: {{ *}}[[@LINE+6]], column: {{[0-9]+}}, indent: 17, spelling: '' }, +// CHECK:STDOUT: ] +// CHECK:STDERR: fail_block_string_second_line.carbon:[[@LINE+4]]:3: Only whitespace is permitted before the closing `'''` of a multi-line string. +// CHECK:STDERR: error here: ''' +// CHECK:STDERR: ^ +var s: String = ''' + error here: ''' diff --git a/toolchain/lexer/tokenized_buffer.cpp b/toolchain/lexer/tokenized_buffer.cpp index 9bd5fdc18223..e466d42c1616 100644 --- a/toolchain/lexer/tokenized_buffer.cpp +++ b/toolchain/lexer/tokenized_buffer.cpp @@ -84,9 +84,9 @@ class TokenizedBuffer::Lexer { Lexer(TokenizedBuffer& buffer, DiagnosticConsumer& consumer) : buffer_(&buffer), - translator_(&buffer, ¤t_column_), + translator_(&buffer), emitter_(translator_, consumer), - token_translator_(&buffer, ¤t_column_), + token_translator_(&buffer), token_emitter_(token_translator_, consumer), current_line_(buffer.AddLine(LineInfo(0))), current_line_info_(&buffer.GetLineInfo(current_line_)) {} @@ -897,8 +897,6 @@ auto TokenizedBuffer::SourceBufferLocationTranslator::GetLocation( const auto* line_it = std::partition_point( buffer_->line_infos_.begin(), buffer_->line_infos_.end(), [offset](const LineInfo& line) { return line.start <= offset; }); - bool incomplete_line_info = last_line_lexed_to_column_ != nullptr && - line_it == buffer_->line_infos_.end(); // Step back one line to find the line containing the given position. CARBON_CHECK(line_it != buffer_->line_infos_.begin()) @@ -907,33 +905,23 @@ auto TokenizedBuffer::SourceBufferLocationTranslator::GetLocation( int line_number = line_it - buffer_->line_infos_.begin(); int column_number = offset - line_it->start; - llvm::StringRef line; - - // We might still be lexing the last line. If so, check to see if there are - // any newline characters between the position we've finished lexing up to - // and the given location. - if (incomplete_line_info && column_number > *last_line_lexed_to_column_) { - column_number = *last_line_lexed_to_column_; - int64_t start = line_it->start; - for (int64_t i = line_it->start + *last_line_lexed_to_column_; i != offset; - ++i) { - if (buffer_->source_->text()[i] == '\n') { - start = i; - ++line_number; - column_number = 0; - } else { - ++column_number; - } + // Start by grabbing the line from the buffer. If the line isn't fully lexed, + // the length will be npos and the line will be grabbed from the known start + // to the end of the buffer; we'll then adjust the length. + llvm::StringRef line = + buffer_->source_->text().substr(line_it->start, line_it->length); + if (line_it->length == static_cast(llvm::StringRef::npos)) { + CARBON_CHECK(line.take_front(column_number).count('\n') == 0) + << "Currently we assume no unlexed newlines prior to the error column, " + "but there was one when erroring at " + << buffer_->source_->filename() << ":" << line_number << ":" + << column_number; + // Look for the next newline since we don't know the length. We can start at + // the column because prior newlines will have been lexed. + auto end_newline_pos = line.find('\n', column_number); + if (end_newline_pos != llvm::StringRef::npos) { + line = line.take_front(end_newline_pos); } - line = buffer_->source_->text().substr(start).take_until( - [](char c) { return c == '\n'; }); - } else if (line_it->length < 0) { - line = - buffer_->source_->text().substr(line_it->start).take_until([](char c) { - return c == '\n'; - }); - } else { - line = buffer_->source_->text().substr(line_it->start, line_it->length); } return {.file_name = buffer_->source_->filename(), @@ -953,8 +941,7 @@ auto TokenizedBuffer::TokenLocationTranslator::GetLocation(Token token) // Find the corresponding file location. // TODO: Should we somehow indicate in the diagnostic location if this token // is a recovery token that doesn't correspond to the original source? - return SourceBufferLocationTranslator(buffer_, last_line_lexed_to_column_) - .GetLocation(token_start); + return SourceBufferLocationTranslator(buffer_).GetLocation(token_start); } } // namespace Carbon diff --git a/toolchain/lexer/tokenized_buffer.h b/toolchain/lexer/tokenized_buffer.h index 3570805ad1f1..dda855c589f6 100644 --- a/toolchain/lexer/tokenized_buffer.h +++ b/toolchain/lexer/tokenized_buffer.h @@ -170,18 +170,14 @@ class TokenizedBuffer { // buffer locations. class TokenLocationTranslator : public DiagnosticLocationTranslator { public: - explicit TokenLocationTranslator(const TokenizedBuffer* buffer, - int* last_line_lexed_to_column) - : buffer_(buffer), - last_line_lexed_to_column_(last_line_lexed_to_column) {} + explicit TokenLocationTranslator(const TokenizedBuffer* buffer) + : buffer_(buffer) {} // Map the given token into a diagnostic location. auto GetLocation(Token token) -> DiagnosticLocation override; private: const TokenizedBuffer* buffer_; - // Passed to SourceBufferLocationTranslator. - int* last_line_lexed_to_column_; }; // Lexes a buffer of source code into a tokenized buffer. @@ -298,10 +294,8 @@ class TokenizedBuffer { class SourceBufferLocationTranslator : public DiagnosticLocationTranslator { public: - explicit SourceBufferLocationTranslator(const TokenizedBuffer* buffer, - int* last_line_lexed_to_column) - : buffer_(buffer), - last_line_lexed_to_column_(last_line_lexed_to_column) {} + explicit SourceBufferLocationTranslator(const TokenizedBuffer* buffer) + : buffer_(buffer) {} // Map the given position within the source buffer into a diagnostic // location. @@ -309,9 +303,6 @@ class TokenizedBuffer { private: const TokenizedBuffer* buffer_; - // The last lexed column, for determining whether the last line should be - // checked for unlexed newlines. May be null after lexing is complete. - int* last_line_lexed_to_column_; }; // Specifies minimum widths to use when printing a token's fields via @@ -360,7 +351,10 @@ class TokenizedBuffer { struct LineInfo { // The length will always be assigned later. Indent may be assigned if // non-zero. - explicit LineInfo(int64_t start) : start(start), length(-1), indent(0) {} + explicit LineInfo(int64_t start) + : start(start), + length(static_cast(llvm::StringRef::npos)), + indent(0) {} // Zero-based byte offset of the start of the line within the source buffer // provided. diff --git a/toolchain/parser/parse_tree.cpp b/toolchain/parser/parse_tree.cpp index 7cee4e069aa0..1321c48d8c41 100644 --- a/toolchain/parser/parse_tree.cpp +++ b/toolchain/parser/parse_tree.cpp @@ -20,8 +20,7 @@ namespace Carbon { auto ParseTree::Parse(TokenizedBuffer& tokens, DiagnosticConsumer& consumer, llvm::raw_ostream* vlog_stream) -> ParseTree { - TokenizedBuffer::TokenLocationTranslator translator( - &tokens, /*last_line_lexed_to_column=*/nullptr); + TokenizedBuffer::TokenLocationTranslator translator(&tokens); TokenDiagnosticEmitter emitter(translator, consumer); // Delegate to the parser. diff --git a/toolchain/parser/parse_tree_node_location_translator.h b/toolchain/parser/parse_tree_node_location_translator.h index cd42c612a8e9..16b9fe4d2084 100644 --- a/toolchain/parser/parse_tree_node_location_translator.h +++ b/toolchain/parser/parse_tree_node_location_translator.h @@ -14,7 +14,7 @@ class ParseTreeNodeLocationTranslator public: explicit ParseTreeNodeLocationTranslator(const TokenizedBuffer* tokens, const ParseTree* parse_tree) - : token_translator_(tokens, nullptr), parse_tree_(parse_tree) {} + : token_translator_(tokens), parse_tree_(parse_tree) {} // Map the given token into a diagnostic location. auto GetLocation(ParseTree::Node node) -> DiagnosticLocation override {