mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-10-05 22:02:55 +01:00
Record //@... directive lines as comments when lexing. (#7494)
The `//@include-in-dumps` and `//@dump-sem-ir-begin`/`-end` tooling directives were consumed for their side effects without a comment record, so the tokens and comments together no longer reconstructed the source: tooling that re-emits a file from them, such as `carbon format`, silently dropped the directive lines. Now each recognized directive line is also recorded as an ordinary full-line comment alongside its side effect. Adjacent full-line comments coalesce into one comment record only within a category, determined by the byte after the `//` introducer: ordinary comments (whitespace, or the end of the line or file), `//@...` directives, and invalid introducers. A transition between categories starts a new record, so a directive next to a comment block is its own comment, while all the invalid spellings lump together to keep the diagnostic noise at one per run. The category boundary also fixes a lost directive: the invalid-comment bulk skip compared only the `//` prefix, so `//!x` directly above `//@dump-sem-ir-begin` absorbed the directive line and its side effect was never recorded. Invalid comment runs now skip line by line (they start from a diagnosed error, so they are not hot) and stop at a whitespace or `@` introducer; the SIMD bulk skip handles only ordinary comment blocks, whose prefix comparison already includes the whitespace byte. Nothing outside the lexer and the formatter reads comment records, and lex dumps do not include comments, so no other behavior changes. Assisted-by: Claude Code
This commit is contained in:
@@ -1163,6 +1163,109 @@ TEST_F(LexerTest, TrailingCommentAfterMultiLineString) {
|
||||
EXPECT_TRUE(buffer.IsTrailingComment(CommentIndex(0)));
|
||||
}
|
||||
|
||||
TEST_F(LexerTest, DirectiveComments) {
|
||||
// A `//@...` directive line is consumed for its tooling side effects and is
|
||||
// also recorded as a comment: the tokens and comments together reconstruct
|
||||
// the source, so tooling such as the formatter preserves the directive.
|
||||
auto& buffer = compile_helper_.GetTokenizedBuffer(
|
||||
"//@include-in-dumps\n"
|
||||
"\n"
|
||||
"//@dump-sem-ir-begin\n"
|
||||
"var x: i32 = 0;\n"
|
||||
"//@dump-sem-ir-end\n");
|
||||
EXPECT_FALSE(buffer.has_errors());
|
||||
EXPECT_TRUE(buffer.has_include_in_dumps());
|
||||
EXPECT_TRUE(buffer.has_dump_sem_ir_ranges());
|
||||
ASSERT_THAT(buffer.comments_size(), Eq(3));
|
||||
EXPECT_THAT(buffer.GetCommentText(CommentIndex(0)),
|
||||
Eq("//@include-in-dumps\n"));
|
||||
EXPECT_THAT(buffer.GetCommentText(CommentIndex(1)),
|
||||
Eq("//@dump-sem-ir-begin\n"));
|
||||
EXPECT_THAT(buffer.GetCommentText(CommentIndex(2)),
|
||||
Eq("//@dump-sem-ir-end\n"));
|
||||
EXPECT_FALSE(buffer.IsTrailingComment(CommentIndex(0)));
|
||||
|
||||
// Adjacent full-line comments coalesce only within a category: a directive
|
||||
// next to an ordinary comment is its own record, while adjacent directives
|
||||
// share one.
|
||||
auto& adjacent = compile_helper_.GetTokenizedBuffer(
|
||||
"// Dump this file.\n"
|
||||
"//@include-in-dumps\n"
|
||||
"//@dump-sem-ir-begin\n"
|
||||
"var x: i32 = 0;\n"
|
||||
"//@dump-sem-ir-end\n");
|
||||
EXPECT_FALSE(adjacent.has_errors());
|
||||
EXPECT_TRUE(adjacent.has_include_in_dumps());
|
||||
ASSERT_THAT(adjacent.comments_size(), Eq(3));
|
||||
EXPECT_THAT(adjacent.GetCommentText(CommentIndex(0)),
|
||||
Eq("// Dump this file.\n"));
|
||||
EXPECT_THAT(adjacent.GetCommentText(CommentIndex(1)),
|
||||
Eq("//@include-in-dumps\n//@dump-sem-ir-begin\n"));
|
||||
EXPECT_THAT(adjacent.GetCommentText(CommentIndex(2)),
|
||||
Eq("//@dump-sem-ir-end\n"));
|
||||
}
|
||||
|
||||
TEST_F(LexerTest, DirectiveAfterInvalidComment) {
|
||||
// An invalid comment line does not absorb a following directive: the
|
||||
// directive ends the invalid run, so its side effects are still recognized
|
||||
// and it is recorded separately.
|
||||
auto& buffer = compile_helper_.GetTokenizedBuffer(
|
||||
"//!bad\n"
|
||||
"//@dump-sem-ir-begin\n"
|
||||
"var x: i32 = 0;\n"
|
||||
"//@dump-sem-ir-end\n");
|
||||
EXPECT_TRUE(buffer.has_errors());
|
||||
EXPECT_TRUE(buffer.has_dump_sem_ir_ranges());
|
||||
ASSERT_THAT(buffer.comments_size(), Eq(3));
|
||||
EXPECT_THAT(buffer.GetCommentText(CommentIndex(0)), Eq("//!bad\n"));
|
||||
EXPECT_THAT(buffer.GetCommentText(CommentIndex(1)),
|
||||
Eq("//@dump-sem-ir-begin\n"));
|
||||
}
|
||||
|
||||
TEST_F(LexerTest, InvalidCommentRunsLumpTogether) {
|
||||
// A run of invalid comment lines lumps into one comment and one diagnostic,
|
||||
// no matter which invalid byte follows each `//`; a valid comment ends the
|
||||
// run and starts its own record.
|
||||
Testing::MockDiagnosticConsumer consumer;
|
||||
EXPECT_CALL(consumer,
|
||||
HandleDiagnostic(IsSingleDiagnostic(
|
||||
Diagnostics::Kind::NoWhitespaceAfterCommentIntroducer,
|
||||
Diagnostics::Level::Error, 1, 3, _)));
|
||||
auto& buffer = compile_helper_.GetTokenizedBuffer(
|
||||
"//!one\n"
|
||||
"//?two\n"
|
||||
"// valid\n",
|
||||
&consumer);
|
||||
ASSERT_THAT(buffer.comments_size(), Eq(2));
|
||||
EXPECT_THAT(buffer.GetCommentText(CommentIndex(0)), Eq("//!one\n//?two\n"));
|
||||
EXPECT_THAT(buffer.GetCommentText(CommentIndex(1)), Eq("// valid\n"));
|
||||
}
|
||||
|
||||
TEST_F(LexerTest, InvalidCommentRunAtEof) {
|
||||
// An invalid run's line-by-line skip reads the byte after each line's `//`;
|
||||
// these cases end the source at that read's boundary, with no trailing
|
||||
// newline, to pin its bounds.
|
||||
|
||||
// The final line's introducer byte is the last byte of the source.
|
||||
auto& tight = compile_helper_.GetTokenizedBuffer(
|
||||
"//!a\n"
|
||||
"//!");
|
||||
EXPECT_TRUE(tight.has_errors());
|
||||
ASSERT_THAT(tight.comments_size(), Eq(1));
|
||||
EXPECT_THAT(tight.GetCommentText(CommentIndex(0)), Eq("//!a\n//!"));
|
||||
|
||||
// A bare `//` ending the source has no byte after the introducer at all:
|
||||
// the run must stop before it rather than read past the end, and it lexes
|
||||
// as its own valid, empty comment.
|
||||
auto& bare = compile_helper_.GetTokenizedBuffer(
|
||||
"//!a\n"
|
||||
"//");
|
||||
EXPECT_TRUE(bare.has_errors());
|
||||
ASSERT_THAT(bare.comments_size(), Eq(2));
|
||||
EXPECT_THAT(bare.GetCommentText(CommentIndex(0)), Eq("//!a\n"));
|
||||
EXPECT_THAT(bare.GetCommentText(CommentIndex(1)), Eq("//"));
|
||||
}
|
||||
|
||||
TEST_F(LexerTest, DiagnosticWhitespace) {
|
||||
Testing::MockDiagnosticConsumer consumer;
|
||||
EXPECT_CALL(consumer,
|
||||
@@ -1318,14 +1421,14 @@ TEST_F(LexerTest, MultipleComments) {
|
||||
{4}
|
||||
x
|
||||
)";
|
||||
constexpr llvm::StringLiteral Comments[] = {
|
||||
constexpr llvm::StringLiteral Groups[] = {
|
||||
// NOLINTNEXTLINE(bugprone-suspicious-missing-comma)
|
||||
"// This comment should be possible to parse with SIMD.\n"
|
||||
"// This one too.\n",
|
||||
"// This one as well, though it's a different indent.\n"
|
||||
" // And mixes indent.\n"
|
||||
" // And mixes indent more.\n",
|
||||
"// This is one comment:\n"
|
||||
"// This is several comments:\n"
|
||||
"//Invalid\n"
|
||||
"// Valid\n"
|
||||
"//Invalid\n"
|
||||
@@ -1334,18 +1437,35 @@ x
|
||||
"//\n"
|
||||
"// Valid\n",
|
||||
"// This uses a high indent, which stops SIMD.\n", "//\n"};
|
||||
std::string source = llvm::formatv(Format, Comments[0], Comments[1],
|
||||
Comments[2], Comments[3], Comments[4])
|
||||
std::string source = llvm::formatv(Format, Groups[0], Groups[1], Groups[2],
|
||||
Groups[3], Groups[4])
|
||||
.str();
|
||||
|
||||
auto& buffer = compile_helper_.GetTokenizedBuffer(source);
|
||||
EXPECT_TRUE(buffer.has_errors());
|
||||
|
||||
EXPECT_THAT(buffer.comments_size(), Eq(std::size(Comments)));
|
||||
for (int i :
|
||||
llvm::seq(std::min<int>(buffer.comments_size(), std::size(Comments)))) {
|
||||
// The third group splits at each transition between ordinary and invalid
|
||||
// comment introducers; the other groups each form one comment.
|
||||
constexpr llvm::StringLiteral ExpectedComments[] = {
|
||||
// NOLINTNEXTLINE(bugprone-suspicious-missing-comma)
|
||||
"// This comment should be possible to parse with SIMD.\n"
|
||||
"// This one too.\n",
|
||||
"// This one as well, though it's a different indent.\n"
|
||||
" // And mixes indent.\n"
|
||||
" // And mixes indent more.\n",
|
||||
"// This is several comments:\n", "//Invalid\n", "// Valid\n",
|
||||
"//Invalid\n",
|
||||
// NOLINTNEXTLINE(bugprone-suspicious-missing-comma)
|
||||
"//\n"
|
||||
"// Valid\n"
|
||||
"//\n"
|
||||
"// Valid\n",
|
||||
"// This uses a high indent, which stops SIMD.\n", "//\n"};
|
||||
EXPECT_THAT(buffer.comments_size(), Eq(std::size(ExpectedComments)));
|
||||
for (int i : llvm::seq(std::min<int>(buffer.comments_size(),
|
||||
std::size(ExpectedComments)))) {
|
||||
EXPECT_THAT(buffer.GetCommentText(CommentIndex(i)).str(),
|
||||
testing::StrEq(Comments[i]));
|
||||
testing::StrEq(ExpectedComments[i]));
|
||||
}
|
||||
EXPECT_THAT(buffer, HasTokens(llvm::ArrayRef<ExpectedToken>{
|
||||
{.kind = TokenKind::FileStart},
|
||||
|
||||
Reference in New Issue
Block a user