From a342e5c117a21e4761c95ca9e6e0bb515bc1494f Mon Sep 17 00:00:00 2001 From: Jon Ross-Perkins Date: Thu, 24 Apr 2025 15:17:04 -0700 Subject: [PATCH] Add lexing for dump-sem-ir-start and end (#5357) Syntax rationale is on `DumpSemIRRange` to try and record this, since I'm not sure this belongs in the language design. The intent of this is to be able to subset SemIR, which will be done separately in the formatter. --------- Co-authored-by: David Blaikie --- toolchain/diagnostics/diagnostic_kind.def | 2 + toolchain/lex/lex.cpp | 107 ++++++++++--- .../lex/testdata/dump_sem_ir_range.carbon | 146 ++++++++++++++++++ toolchain/lex/tokenized_buffer.cpp | 8 + toolchain/lex/tokenized_buffer.h | 19 +++ 5 files changed, 262 insertions(+), 20 deletions(-) create mode 100644 toolchain/lex/testdata/dump_sem_ir_range.carbon diff --git a/toolchain/diagnostics/diagnostic_kind.def b/toolchain/diagnostics/diagnostic_kind.def index 697a92d9e5e8..3866e5fe1641 100644 --- a/toolchain/diagnostics/diagnostic_kind.def +++ b/toolchain/diagnostics/diagnostic_kind.def @@ -46,6 +46,8 @@ CARBON_DIAGNOSTIC_KIND(ErrorReadingFile) CARBON_DIAGNOSTIC_KIND(BinaryRealLiteral) CARBON_DIAGNOSTIC_KIND(ContentBeforeStringTerminator) CARBON_DIAGNOSTIC_KIND(DecimalEscapeSequence) +CARBON_DIAGNOSTIC_KIND(DumpSemIRRangeMissingEnd) +CARBON_DIAGNOSTIC_KIND(DumpSemIRRangeMissingStart) CARBON_DIAGNOSTIC_KIND(EmptyDigitSequence) CARBON_DIAGNOSTIC_KIND(HexadecimalEscapeMissingDigits) CARBON_DIAGNOSTIC_KIND(InvalidDigit) diff --git a/toolchain/lex/lex.cpp b/toolchain/lex/lex.cpp index 91bbe0aea372..1915b05e88ca 100644 --- a/toolchain/lex/lex.cpp +++ b/toolchain/lex/lex.cpp @@ -145,6 +145,10 @@ class [[clang::internal_linkage]] Lexer { auto SkipHorizontalWhitespace(llvm::StringRef source_text, ssize_t& position) -> void; + // Starts a new line, skipping whitespace and setting the indent. + auto AdvanceToLine(llvm::StringRef source_text, ssize_t& position, + ssize_t to_line_index) -> void; + auto LexHorizontalWhitespace(llvm::StringRef source_text, ssize_t& position) -> void; @@ -208,9 +212,19 @@ class [[clang::internal_linkage]] Lexer { // should always fully consume the source text. auto Lex() && -> TokenizedBuffer; + // Checks for an ends a `DumpSemIRRange` that's missing an explicit end + // marker. + auto EndDumpSemIRRangeIfIncomplete(const char* diag_loc) -> void; + private: class ErrorRecoveryBuffer; + // Handles `//@dump-sem-ir-start` for a `DumpSemIRRange`. + auto StartDumpSemIRRange(const char* diag_loc) -> void; + + // Handles `//@dump-sem-ir-end` for a `DumpSemIRRange`. + auto EndDumpSemIRRange(const char* diag_loc) -> void; + TokenizedBuffer buffer_; ssize_t line_index_; @@ -669,6 +683,11 @@ static auto DispatchNext(Lexer& lexer, llvm::StringRef source_text, source_text[position])](lexer, source_text, position); } + // Incomplete ranges will use the next token for their end; we want that to be + // `FileEnd` in this case, so check before adding `FileEnd`. The argument is + // just the final character for diagnostic locations. + lexer.EndDumpSemIRRangeIfIncomplete(source_text.end() - 1); + // When we finish the source text, stop recursing. We also hint this so that // the tail-dispatch is optimized as that's essentially the loop back-edge // and this is the loop exit. @@ -804,6 +823,17 @@ auto Lexer::SkipHorizontalWhitespace(llvm::StringRef source_text, } } +auto Lexer::AdvanceToLine(llvm::StringRef source_text, ssize_t& position, + ssize_t to_line_index) -> void { + CARBON_DCHECK(to_line_index >= line_index_); + line_index_ = to_line_index; + auto* line_info = current_line_info(); + ssize_t line_start = line_info->start; + position = line_start; + SkipHorizontalWhitespace(source_text, position); + line_info->indent = position - line_start; +} + auto Lexer::LexHorizontalWhitespace(llvm::StringRef source_text, ssize_t& position) -> void { CARBON_DCHECK(source_text[position] == ' ' || source_text[position] == '\t'); @@ -815,12 +845,7 @@ auto Lexer::LexHorizontalWhitespace(llvm::StringRef source_text, auto Lexer::LexVerticalWhitespace(llvm::StringRef source_text, ssize_t& position) -> void { NoteWhitespace(); - ++line_index_; - auto* line_info = current_line_info(); - ssize_t line_start = line_info->start; - position = line_start; - SkipHorizontalWhitespace(source_text, position); - line_info->indent = position - line_start; + AdvanceToLine(source_text, position, line_index_ + 1); } auto Lexer::LexCR(llvm::StringRef source_text, ssize_t& position) -> void { @@ -870,6 +895,46 @@ auto Lexer::LexCommentOrSlash(llvm::StringRef source_text, ssize_t& position) CARBON_CHECK(result, "Failed to form a token!"); } +auto Lexer::StartDumpSemIRRange(const char* diag_loc) -> void { + EndDumpSemIRRangeIfIncomplete(diag_loc); + + // The start here will be the next token, which may be FileEnd. The end will + // be assigned by either AddDumpSemIREnd or, if invalid, + // EndDumpSemIRRangeIfIncomplete. + buffer_.dump_sem_ir_ranges_.push_back( + {.start = TokenIndex(buffer_.size()), .end = TokenIndex::None}); +} + +auto Lexer::EndDumpSemIRRange(const char* diag_loc) -> void { + if (buffer_.dump_sem_ir_ranges_.empty() || + buffer_.dump_sem_ir_ranges_.back().end != TokenIndex::None) { + CARBON_DIAGNOSTIC( + DumpSemIRRangeMissingStart, Error, + "missing `//@dump-sem-ir-start` to match `//@dump-sem-ir-end`"); + emitter_.Emit(diag_loc, DumpSemIRRangeMissingStart); + return; + } + + buffer_.dump_sem_ir_ranges_.back().end = TokenIndex(buffer_.size()); +} + +auto Lexer::EndDumpSemIRRangeIfIncomplete(const char* diag_loc) -> void { + if (buffer_.dump_sem_ir_ranges_.empty() || + buffer_.dump_sem_ir_ranges_.back().end != TokenIndex::None) { + return; + } + + // The location here won't be closely associated with the start location. + // However, this is a developer feature and not worth complexity to diagnose + // better. + CARBON_DIAGNOSTIC( + DumpSemIRRangeMissingEnd, Error, + "missing `//@dump-sem-ir-end` to match `//@dump-sem-ir-start`"); + emitter_.Emit(diag_loc, DumpSemIRRangeMissingEnd); + + EndDumpSemIRRange(diag_loc); +} + auto Lexer::LexComment(llvm::StringRef source_text, ssize_t& position) -> void { CARBON_DCHECK(source_text.substr(position).starts_with("//")); int32_t comment_start = position; @@ -895,10 +960,21 @@ auto Lexer::LexComment(llvm::StringRef source_text, ssize_t& position) -> void { bool is_valid_after_slashes = true; if (position + 2 < static_cast(source_text.size()) && LLVM_UNLIKELY(!IsSpace(source_text[position + 2]))) { + llvm::StringRef comment_text = source_text.substr(position); + if (comment_text.starts_with("//@dump-sem-ir-start\n")) { + StartDumpSemIRRange(comment_text.begin()); + AdvanceToLine(source_text, position, line_index_ + 1); + return; + } + if (comment_text.starts_with("//@dump-sem-ir-end\n")) { + EndDumpSemIRRange(comment_text.begin()); + AdvanceToLine(source_text, position, line_index_ + 1); + return; + } + CARBON_DIAGNOSTIC(NoWhitespaceAfterCommentIntroducer, Error, "whitespace is required after '//'"); - emitter_.Emit(source_text.begin() + position + 2, - NoWhitespaceAfterCommentIntroducer); + emitter_.Emit(comment_text.begin() + 2, NoWhitespaceAfterCommentIntroducer); // We use this to tweak the lexing of blocks below. is_valid_after_slashes = false; @@ -992,14 +1068,7 @@ auto Lexer::LexComment(llvm::StringRef source_text, ssize_t& position) -> void { } buffer_.AddComment(indent, comment_start, position); - - // Now compute the indent of this next line before we finish. - ssize_t line_start = position; - SkipHorizontalWhitespace(source_text, position); - - // Now that we're done scanning, update to the latest line index and indent. - line_index_ = line_index; - current_line_info()->indent = position - line_start; + AdvanceToLine(source_text, position, line_index); } auto Lexer::CanFormRealLiteral() -> bool { @@ -1373,10 +1442,8 @@ auto Lexer::LexFileStart(llvm::StringRef source_text, ssize_t& position) // Also skip any horizontal whitespace and record the indentation of the // first line. - SkipHorizontalWhitespace(source_text, position); - auto* line_info = current_line_info(); - CARBON_CHECK(line_info->start == 0); - line_info->indent = position; + CARBON_CHECK(current_line_info()->start == 0); + AdvanceToLine(source_text, position, /*to_line_index=*/0); } auto Lexer::LexFileEnd(llvm::StringRef source_text, ssize_t position) -> void { diff --git a/toolchain/lex/testdata/dump_sem_ir_range.carbon b/toolchain/lex/testdata/dump_sem_ir_range.carbon new file mode 100644 index 000000000000..f38e82363858 --- /dev/null +++ b/toolchain/lex/testdata/dump_sem_ir_range.carbon @@ -0,0 +1,146 @@ +// Part of the Carbon Language project, under the Apache License v2.0 with LLVM +// Exceptions. See /LICENSE for license information. +// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception +// +// AUTOUPDATE +// TIP: To test this file alone, run: +// TIP: bazel test //toolchain/testing:file_test --test_arg=--file_tests=toolchain/lex/testdata/dump_sem_ir_range.carbon +// TIP: To dump output, run: +// TIP: bazel run //toolchain/testing:file_test -- --dump_output --file_tests=toolchain/lex/testdata/dump_sem_ir_range.carbon + +// --- full_file.carbon +// CHECK:STDOUT: - filename: full_file.carbon +// CHECK:STDOUT: tokens: + +//@dump-sem-ir-start +a +// CHECK:STDOUT: - { index: 1, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "a", identifier: 0, has_leading_space: true } +b +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "b", identifier: 1, has_leading_space: true } +c +// CHECK:STDOUT: - { index: 3, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "c", identifier: 2, has_leading_space: true } +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 1, end: 4} +//@dump-sem-ir-end + +// --- multi_section.carbon +// CHECK:STDOUT: - filename: multi_section.carbon +// CHECK:STDOUT: tokens: + +a +// CHECK:STDOUT: - { index: 1, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "a", identifier: 0, has_leading_space: true } +//@dump-sem-ir-start +b +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "b", identifier: 1, has_leading_space: true } +//@dump-sem-ir-end +c +// CHECK:STDOUT: - { index: 3, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "c", identifier: 2, has_leading_space: true } +//@dump-sem-ir-start +d +// CHECK:STDOUT: - { index: 4, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "d", identifier: 3, has_leading_space: true } +//@dump-sem-ir-end +e +// CHECK:STDOUT: - { index: 5, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "e", identifier: 4, has_leading_space: true } +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 2, end: 3} +// CHECK:STDOUT: - {start: 4, end: 5} + +// --- compact.carbon +// CHECK:STDOUT: - filename: compact.carbon +// CHECK:STDOUT: tokens: +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 1, end: 1} +// CHECK:STDOUT: - {start: 1, end: 1} +// CHECK:STDOUT: - {start: 1, end: 1} +//@dump-sem-ir-start +//@dump-sem-ir-end +//@dump-sem-ir-start +//@dump-sem-ir-end +//@dump-sem-ir-start +//@dump-sem-ir-end + +// --- fail_extra_text.carbon +// CHECK:STDOUT: - filename: fail_extra_text.carbon +// CHECK:STDOUT: tokens: + +// CHECK:STDERR: fail_extra_text.carbon:[[@LINE+4]]:3: error: whitespace is required after '//' [NoWhitespaceAfterCommentIntroducer] +// CHECK:STDERR: //@dump-sem-ir-start more text +// CHECK:STDERR: ^ +// CHECK:STDERR: +//@dump-sem-ir-start more text + +// --- fail_start_only.carbon +// CHECK:STDOUT: - filename: fail_start_only.carbon +// CHECK:STDOUT: tokens: +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 1, end: 1} + +//@dump-sem-ir-start +// CHECK:STDERR: fail_start_only.carbon:[[@LINE+4]]:1: error: missing `//@dump-sem-ir-end` to match `//@dump-sem-ir-start` [DumpSemIRRangeMissingEnd] +// CHECK:STDERR: +// CHECK:STDERR: ^ +// CHECK:STDERR: + +// --- fail_end_only.carbon +// CHECK:STDOUT: - filename: fail_end_only.carbon +// CHECK:STDOUT: tokens: + +// CHECK:STDERR: fail_end_only.carbon:[[@LINE+4]]:1: error: missing `//@dump-sem-ir-start` to match `//@dump-sem-ir-end` [DumpSemIRRangeMissingStart] +// CHECK:STDERR: //@dump-sem-ir-end +// CHECK:STDERR: ^ +// CHECK:STDERR: +//@dump-sem-ir-end + +// --- fail_misordered.carbon +// CHECK:STDOUT: - filename: fail_misordered.carbon +// CHECK:STDOUT: tokens: + +a +// CHECK:STDOUT: - { index: 1, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "a", identifier: 0, has_leading_space: true } +//@dump-sem-ir-start +b +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "b", identifier: 1, has_leading_space: true } +// CHECK:STDERR: fail_misordered.carbon:[[@LINE+4]]:1: error: missing `//@dump-sem-ir-end` to match `//@dump-sem-ir-start` [DumpSemIRRangeMissingEnd] +// CHECK:STDERR: //@dump-sem-ir-start +// CHECK:STDERR: ^ +// CHECK:STDERR: +//@dump-sem-ir-start +c +// CHECK:STDOUT: - { index: 3, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "c", identifier: 2, has_leading_space: true } +//@dump-sem-ir-end +d +// CHECK:STDOUT: - { index: 4, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "d", identifier: 3, has_leading_space: true } +// CHECK:STDERR: fail_misordered.carbon:[[@LINE+4]]:1: error: missing `//@dump-sem-ir-start` to match `//@dump-sem-ir-end` [DumpSemIRRangeMissingStart] +// CHECK:STDERR: //@dump-sem-ir-end +// CHECK:STDERR: ^ +// CHECK:STDERR: +//@dump-sem-ir-end +e +// CHECK:STDOUT: - { index: 5, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "e", identifier: 4, has_leading_space: true } +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 2, end: 3} +// CHECK:STDOUT: - {start: 3, end: 4} + +// --- fail_count_mismatch.carbon +// CHECK:STDOUT: - filename: fail_count_mismatch.carbon +// CHECK:STDOUT: tokens: + +a +// CHECK:STDOUT: - { index: 1, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "a", identifier: 0, has_leading_space: true } +//@dump-sem-ir-start +b +// CHECK:STDOUT: - { index: 2, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "b", identifier: 1, has_leading_space: true } +//@dump-sem-ir-end +c +// CHECK:STDOUT: - { index: 3, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "c", identifier: 2, has_leading_space: true } +//@dump-sem-ir-start +d +// CHECK:STDOUT: - { index: 4, kind: "Identifier", line: {{ *}}[[@LINE-1]], column: 1, indent: 1, spelling: "d", identifier: 3, has_leading_space: true } +// CHECK:STDOUT: dump_sem_ir_ranges: +// CHECK:STDOUT: - {start: 2, end: 3} +// CHECK:STDOUT: - {start: 4, end: 5} + +// CHECK:STDERR: fail_count_mismatch.carbon:[[@LINE+3]]:17: error: missing `//@dump-sem-ir-end` to match `//@dump-sem-ir-start` [DumpSemIRRangeMissingEnd] +// CHECK:STDERR: // CHECK:STDERR: +// CHECK:STDERR: ^ +// CHECK:STDERR: diff --git a/toolchain/lex/tokenized_buffer.cpp b/toolchain/lex/tokenized_buffer.cpp index 6a49128628e8..428a55c8412b 100644 --- a/toolchain/lex/tokenized_buffer.cpp +++ b/toolchain/lex/tokenized_buffer.cpp @@ -236,6 +236,14 @@ auto TokenizedBuffer::Print(llvm::raw_ostream& output_stream, PrintToken(output_stream, token, widths); output_stream << "\n"; } + + if (!dump_sem_ir_ranges_.empty()) { + output_stream << " dump_sem_ir_ranges:\n"; + for (auto range : dump_sem_ir_ranges_) { + output_stream << " - {start: " << range.start.index + << ", end: " << range.end.index << "}\n"; + } + } } auto TokenizedBuffer::PrintToken(llvm::raw_ostream& output_stream, diff --git a/toolchain/lex/tokenized_buffer.h b/toolchain/lex/tokenized_buffer.h index e91db739d147..41dbd3b5df3d 100644 --- a/toolchain/lex/tokenized_buffer.h +++ b/toolchain/lex/tokenized_buffer.h @@ -85,6 +85,18 @@ class TokenizedBuffer : public Printable { LineIndex start_line; }; + // A range of tokens marked by `//@dump-semir-[start|end]`. The end token is + // non-inclusive: [start, end). + // + // The particular syntax was chosen because it can be lexed efficiently. It + // only occurs in invalid comment strings, so shouldn't slow down lexing of + // correct code. It's also comment-like because its presence won't affect + // parse/check. + struct DumpSemIRRange { + TokenIndex start; + TokenIndex end; + }; + auto GetKind(TokenIndex token) const -> TokenKind; auto GetLine(TokenIndex token) const -> LineIndex; @@ -197,6 +209,10 @@ class TokenizedBuffer : public Printable { auto comments_size() const -> size_t { return comments_.size(); } + auto dump_sem_ir_ranges() -> llvm::ArrayRef { + return dump_sem_ir_ranges_; + } + // This is an upper bound on the number of output parse nodes in the absence // of errors. auto expected_max_parse_tree_size() const -> int { @@ -482,6 +498,9 @@ class TokenizedBuffer : public Printable { // Comments in the file. llvm::SmallVector comments_; + // Ranges of SemIR to dump. + llvm::SmallVector dump_sem_ir_ranges_; + // An upper bound on the number of parse tree nodes that we expect to be // created for the tokens in this buffer. int expected_max_parse_tree_size_ = 0;