diff --git a/toolchain/diagnostics/diagnostic_kind.def b/toolchain/diagnostics/diagnostic_kind.def index 697a92d9e5e8..3866e5fe1641 100644 --- a/toolchain/diagnostics/diagnostic_kind.def +++ b/toolchain/diagnostics/diagnostic_kind.def @@ -46,6 +46,8 @@ CARBON_DIAGNOSTIC_KIND(ErrorReadingFile) CARBON_DIAGNOSTIC_KIND(BinaryRealLiteral) CARBON_DIAGNOSTIC_KIND(ContentBeforeStringTerminator) CARBON_DIAGNOSTIC_KIND(DecimalEscapeSequence) +CARBON_DIAGNOSTIC_KIND(DumpSemIRRangeMissingEnd) +CARBON_DIAGNOSTIC_KIND(DumpSemIRRangeMissingStart) CARBON_DIAGNOSTIC_KIND(EmptyDigitSequence) CARBON_DIAGNOSTIC_KIND(HexadecimalEscapeMissingDigits) CARBON_DIAGNOSTIC_KIND(InvalidDigit) diff --git a/toolchain/lex/lex.cpp b/toolchain/lex/lex.cpp index 91bbe0aea372..1915b05e88ca 100644 --- a/toolchain/lex/lex.cpp +++ b/toolchain/lex/lex.cpp @@ -145,6 +145,10 @@ class [[clang::internal_linkage]] Lexer { auto SkipHorizontalWhitespace(llvm::StringRef source_text, ssize_t& position) -> void; + // Starts a new line, skipping whitespace and setting the indent. + auto AdvanceToLine(llvm::StringRef source_text, ssize_t& position, + ssize_t to_line_index) -> void; + auto LexHorizontalWhitespace(llvm::StringRef source_text, ssize_t& position) -> void; @@ -208,9 +212,19 @@ class [[clang::internal_linkage]] Lexer { // should always fully consume the source text. auto Lex() && -> TokenizedBuffer; + // Checks for an ends a `DumpSemIRRange` that's missing an explicit end + // marker. + auto EndDumpSemIRRangeIfIncomplete(const char* diag_loc) -> void; + private: class ErrorRecoveryBuffer; + // Handles `//@dump-sem-ir-start` for a `DumpSemIRRange`. + auto StartDumpSemIRRange(const char* diag_loc) -> void; + + // Handles `//@dump-sem-ir-end` for a `DumpSemIRRange`. + auto EndDumpSemIRRange(const char* diag_loc) -> void; + TokenizedBuffer buffer_; ssize_t line_index_; @@ -669,6 +683,11 @@ static auto DispatchNext(Lexer& lexer, llvm::StringRef source_text, source_text[position])](lexer, source_text, position); } + // Incomplete ranges will use the next token for their end; we want that to be + // `FileEnd` in this case, so check before adding `FileEnd`. The argument is + // just the final character for diagnostic locations. + lexer.EndDumpSemIRRangeIfIncomplete(source_text.end() - 1); + // When we finish the source text, stop recursing. We also hint this so that // the tail-dispatch is optimized as that's essentially the loop back-edge // and this is the loop exit. @@ -804,6 +823,17 @@ auto Lexer::SkipHorizontalWhitespace(llvm::StringRef source_text, } } +auto Lexer::AdvanceToLine(llvm::StringRef source_text, ssize_t& position, + ssize_t to_line_index) -> void { + CARBON_DCHECK(to_line_index >= line_index_); + line_index_ = to_line_index; + auto* line_info = current_line_info(); + ssize_t line_start = line_info->start; + position = line_start; + SkipHorizontalWhitespace(source_text, position); + line_info->indent = position - line_start; +} + auto Lexer::LexHorizontalWhitespace(llvm::StringRef source_text, ssize_t& position) -> void { CARBON_DCHECK(source_text[position] == ' ' || source_text[position] == '\t'); @@ -815,12 +845,7 @@ auto Lexer::LexHorizontalWhitespace(llvm::StringRef source_text, auto Lexer::LexVerticalWhitespace(llvm::StringRef source_text, ssize_t& position) -> void { NoteWhitespace(); - ++line_index_; - auto* line_info = current_line_info(); - ssize_t line_start = line_info->start; - position = line_start; - SkipHorizontalWhitespace(source_text, position); - line_info->indent = position - line_start; + AdvanceToLine(source_text, position, line_index_ + 1); } auto Lexer::LexCR(llvm::StringRef source_text, ssize_t& position) -> void { @@ -870,6 +895,46 @@ auto Lexer::LexCommentOrSlash(llvm::StringRef source_text, ssize_t& position) CARBON_CHECK(result, "Failed to form a token!"); } +auto Lexer::StartDumpSemIRRange(const char* diag_loc) -> void { + EndDumpSemIRRangeIfIncomplete(diag_loc); + + // The start here will be the next token, which may be FileEnd. The end will + // be assigned by either AddDumpSemIREnd or, if invalid, + // EndDumpSemIRRangeIfIncomplete. + buffer_.dump_sem_ir_ranges_.push_back( + {.start = TokenIndex(buffer_.size()), .end = TokenIndex::None}); +} + +auto Lexer::EndDumpSemIRRange(const char* diag_loc) -> void { + if (buffer_.dump_sem_ir_ranges_.empty() || + buffer_.dump_sem_ir_ranges_.back().end != TokenIndex::None) { + CARBON_DIAGNOSTIC( + DumpSemIRRangeMissingStart, Error, + "missing `//@dump-sem-ir-start` to match `//@dump-sem-ir-end`"); + emitter_.Emit(diag_loc, DumpSemIRRangeMissingStart); + return; + } + + buffer_.dump_sem_ir_ranges_.back().end = TokenIndex(buffer_.size()); +} + +auto Lexer::EndDumpSemIRRangeIfIncomplete(const char* diag_loc) -> void { + if (buffer_.dump_sem_ir_ranges_.empty() || + buffer_.dump_sem_ir_ranges_.back().end != TokenIndex::None) { + return; + } + + // The location here won't be closely associated with the start location. + // However, this is a developer feature and not worth complexity to diagnose + // better. + CARBON_DIAGNOSTIC( + DumpSemIRRangeMissingEnd, Error, + "missing `//@dump-sem-ir-end` to match `//@dump-sem-ir-start`"); + emitter_.Emit(diag_loc, DumpSemIRRangeMissingEnd); + + EndDumpSemIRRange(diag_loc); +} + auto Lexer::LexComment(llvm::StringRef source_text, ssize_t& position) -> void { CARBON_DCHECK(source_text.substr(position).starts_with("//")); int32_t comment_start = position; @@ -895,10 +960,21 @@ auto Lexer::LexComment(llvm::StringRef source_text, ssize_t& position) -> void { bool is_valid_after_slashes = true; if (position + 2 < static_cast(source_text.size()) && LLVM_UNLIKELY(!IsSpace(source_text[position + 2]))) { + llvm::StringRef comment_text = source_text.substr(position); + if (comment_text.starts_with("//@dump-sem-ir-start\n")) { + StartDumpSemIRRange(comment_text.begin()); + AdvanceToLine(source_text, position, line_index_ + 1); + return; + } + if (comment_text.starts_with("//@dump-sem-ir-end\n")) { + EndDumpSemIRRange(comment_text.begin()); + AdvanceToLine(source_text, position, line_index_ + 1); + return; + } + CARBON_DIAGNOSTIC(NoWhitespaceAfterCommentIntroducer, Error, "whitespace is required after '//'"); - emitter_.Emit(source_text.begin() + position + 2, - NoWhitespaceAfterCommentIntroducer); + emitter_.Emit(comment_text.begin() + 2, NoWhitespaceAfterCommentIntroducer); // We use this to tweak the lexing of blocks below. is_valid_after_slashes = false; @@ -992,14 +1068,7 @@ auto Lexer::LexComment(llvm::StringRef source_text, ssize_t& position) -> void { } buffer_.AddComment(indent, comment_start, position); - - // Now compute the indent of this next line before we finish. - ssize_t line_start = position; - SkipHorizontalWhitespace(source_text, position); - - // Now that we're done scanning, update to the latest line index and indent. - line_index_ = line_index; - current_line_info()->indent = position - line_start; + AdvanceToLine(source_text, position, line_index); } auto Lexer::CanFormRealLiteral() -> bool { @@ -1373,10 +1442,8 @@ auto Lexer::LexFileStart(llvm::StringRef source_text, ssize_t& position) // Also skip any horizontal whitespace and record the indentation of the // first line. - SkipHorizontalWhitespace(source_text, position); - auto* line_info = current_line_info(); - CARBON_CHECK(line_info->start == 0); - line_info->indent = position; + CARBON_CHECK(current_line_info()->start == 0); + AdvanceToLine(source_text, position, /*to_line_index=*/0); } auto Lexer::LexFileEnd(llvm::StringRef source_text, ssize_t position) -> void { diff --git a/toolchain/lex/testdata/dump_sem_ir_range.carbon b/toolchain/lex/testdata/dump_sem_ir_range.carbon new file mode 100644 index 000000000000..f38e82363858 --- /dev/null +++ b/toolchain/lex/testdata/dump_sem_ir_range.carbon @@ -0,0 +1,146 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +// AUTOUPDATE +// TIP: To test this file alone, run: +// TIP: bazel test //toolchain/testing:file_test --test_arg=--file_tests=toolchain/lex/testdata/dump_sem_ir_range.carbon +// TIP: To dump output, run: +// TIP: bazel run //toolchain/testing:file_test -- --dump_output --file_tests=toolchain/lex/testdata/dump_sem_ir_range.carbon + +// --- full_file.carbon +// CHECK:STDOUT: - filename: full_file.carbon +// CHECK:STDOUT: tokens: + +//@dump-sem-ir-start +a +// CHECK:STDOUT: - { index: 1, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "a", identifier: 0, has_leading_space: true } +b +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "b", identifier: 1, has_leading_space: true } +c +// CHECK:STDOUT: - { index: 3, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "c", identifier: 2, has_leading_space: true } +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 1, end: 4} +//@dump-sem-ir-end + +// --- multi_section.carbon +// CHECK:STDOUT: - filename: multi_section.carbon +// CHECK:STDOUT: tokens: + +a +// CHECK:STDOUT: - { index: 1, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "a", identifier: 0, has_leading_space: true } +//@dump-sem-ir-start +b +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "b", identifier: 1, has_leading_space: true } +//@dump-sem-ir-end +c +// CHECK:STDOUT: - { index: 3, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "c", identifier: 2, has_leading_space: true } +//@dump-sem-ir-start +d +// CHECK:STDOUT: - { index: 4, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "d", identifier: 3, has_leading_space: true } +//@dump-sem-ir-end +e +// CHECK:STDOUT: - { index: 5, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "e", identifier: 4, has_leading_space: true } +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 2, end: 3} +// CHECK:STDOUT: - {start: 4, end: 5} + +// --- compact.carbon +// CHECK:STDOUT: - filename: compact.carbon +// CHECK:STDOUT: tokens: +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 1, end: 1} +// CHECK:STDOUT: - {start: 1, end: 1} +// CHECK:STDOUT: - {start: 1, end: 1} +//@dump-sem-ir-start +//@dump-sem-ir-end +//@dump-sem-ir-start +//@dump-sem-ir-end +//@dump-sem-ir-start +//@dump-sem-ir-end + +// --- fail_extra_text.carbon +// CHECK:STDOUT: - filename: fail_extra_text.carbon +// CHECK:STDOUT: tokens: + +// CHECK:STDERR: fail_extra_text.carbon:[[@LINE+4]]:3: error: whitespace is required after '//' [NoWhitespaceAfterCommentIntroducer] +// CHECK:STDERR: //@dump-sem-ir-start more text +// CHECK:STDERR: ^ +// CHECK:STDERR: +//@dump-sem-ir-start more text + +// --- fail_start_only.carbon +// CHECK:STDOUT: - filename: fail_start_only.carbon +// CHECK:STDOUT: tokens: +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 1, end: 1} + +//@dump-sem-ir-start +// CHECK:STDERR: fail_start_only.carbon:[[@LINE+4]]:1: error: missing `//@dump-sem-ir-end` to match `//@dump-sem-ir-start` [DumpSemIRRangeMissingEnd] +// CHECK:STDERR: +// CHECK:STDERR: ^ +// CHECK:STDERR: + +// --- fail_end_only.carbon +// CHECK:STDOUT: - filename: fail_end_only.carbon +// CHECK:STDOUT: tokens: + +// CHECK:STDERR: fail_end_only.carbon:[[@LINE+4]]:1: error: missing `//@dump-sem-ir-start` to match `//@dump-sem-ir-end` [DumpSemIRRangeMissingStart] +// CHECK:STDERR: //@dump-sem-ir-end +// CHECK:STDERR: ^ +// CHECK:STDERR: +//@dump-sem-ir-end + +// --- fail_misordered.carbon +// CHECK:STDOUT: - filename: fail_misordered.carbon +// CHECK:STDOUT: tokens: + +a +// CHECK:STDOUT: - { index: 1, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "a", identifier: 0, has_leading_space: true } +//@dump-sem-ir-start +b +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "b", identifier: 1, has_leading_space: true } +// CHECK:STDERR: fail_misordered.carbon:[[@LINE+4]]:1: error: missing `//@dump-sem-ir-end` to match `//@dump-sem-ir-start` [DumpSemIRRangeMissingEnd] +// CHECK:STDERR: //@dump-sem-ir-start +// CHECK:STDERR: ^ +// CHECK:STDERR: +//@dump-sem-ir-start +c +// CHECK:STDOUT: - { index: 3, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "c", identifier: 2, has_leading_space: true } +//@dump-sem-ir-end +d +// CHECK:STDOUT: - { index: 4, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "d", identifier: 3, has_leading_space: true } +// CHECK:STDERR: fail_misordered.carbon:[[@LINE+4]]:1: error: missing `//@dump-sem-ir-start` to match `//@dump-sem-ir-end` [DumpSemIRRangeMissingStart] +// CHECK:STDERR: //@dump-sem-ir-end +// CHECK:STDERR: ^ +// CHECK:STDERR: +//@dump-sem-ir-end +e +// CHECK:STDOUT: - { index: 5, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "e", identifier: 4, has_leading_space: true } +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 2, end: 3} +// CHECK:STDOUT: - {start: 3, end: 4} + +// --- fail_count_mismatch.carbon +// CHECK:STDOUT: - filename: fail_count_mismatch.carbon +// CHECK:STDOUT: tokens: + +a +// CHECK:STDOUT: - { index: 1, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "a", identifier: 0, has_leading_space: true } +//@dump-sem-ir-start +b +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "b", identifier: 1, has_leading_space: true } +//@dump-sem-ir-end +c +// CHECK:STDOUT: - { index: 3, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "c", identifier: 2, has_leading_space: true } +//@dump-sem-ir-start +d +// CHECK:STDOUT: - { index: 4, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "d", identifier: 3, has_leading_space: true } +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 2, end: 3} +// CHECK:STDOUT: - {start: 4, end: 5} + +// CHECK:STDERR: fail_count_mismatch.carbon:[[@LINE+3]]:17: error: missing `//@dump-sem-ir-end` to match `//@dump-sem-ir-start` [DumpSemIRRangeMissingEnd] +// CHECK:STDERR: // CHECK:STDERR: +// CHECK:STDERR: ^ +// CHECK:STDERR: diff --git a/toolchain/lex/tokenized_buffer.cpp b/toolchain/lex/tokenized_buffer.cpp index 6a49128628e8..428a55c8412b 100644 --- a/toolchain/lex/tokenized_buffer.cpp +++ b/toolchain/lex/tokenized_buffer.cpp @@ -236,6 +236,14 @@ auto TokenizedBuffer::Print(llvm::raw_ostream& output_stream, PrintToken(output_stream, token, widths); output_stream << "\n"; } + + if (!dump_sem_ir_ranges_.empty()) { + output_stream << " dump_sem_ir_ranges:\n"; + for (auto range : dump_sem_ir_ranges_) { + output_stream << " - {start: " << range.start.index + << ", end: " << range.end.index << "}\n"; + } + } } auto TokenizedBuffer::PrintToken(llvm::raw_ostream& output_stream, diff --git a/toolchain/lex/tokenized_buffer.h b/toolchain/lex/tokenized_buffer.h index e91db739d147..41dbd3b5df3d 100644 --- a/toolchain/lex/tokenized_buffer.h +++ b/toolchain/lex/tokenized_buffer.h @@ -85,6 +85,18 @@ class TokenizedBuffer : public Printable { LineIndex start_line; }; + // A range of tokens marked by `//@dump-semir-[start|end]`. The end token is + // non-inclusive: [start, end). + // + // The particular syntax was chosen because it can be lexed efficiently. It + // only occurs in invalid comment strings, so shouldn't slow down lexing of + // correct code. It's also comment-like because its presence won't affect + // parse/check. + struct DumpSemIRRange { + TokenIndex start; + TokenIndex end; + }; + auto GetKind(TokenIndex token) const -> TokenKind; auto GetLine(TokenIndex token) const -> LineIndex; @@ -197,6 +209,10 @@ class TokenizedBuffer : public Printable { auto comments_size() const -> size_t { return comments_.size(); } + auto dump_sem_ir_ranges() -> llvm::ArrayRef { + return dump_sem_ir_ranges_; + } + // This is an upper bound on the number of output parse nodes in the absence // of errors. auto expected_max_parse_tree_size() const -> int { @@ -482,6 +498,9 @@ class TokenizedBuffer : public Printable { // Comments in the file. llvm::SmallVector comments_; + // Ranges of SemIR to dump. + llvm::SmallVector dump_sem_ir_ranges_; + // An upper bound on the number of parse tree nodes that we expect to be // created for the tokens in this buffer. int expected_max_parse_tree_size_ = 0;