Record //@... directive lines as comments when lexing. (#7494)

The `//@include-in-dumps` and `//@dump-sem-ir-begin`/`-end` tooling
directives were consumed for their side effects without a comment
record, so the tokens and comments together no longer reconstructed the
source: tooling that re-emits a file from them, such as `carbon format`,
silently dropped the directive lines. Now each recognized directive line
is also recorded as an ordinary full-line comment alongside its side
effect.

Adjacent full-line comments coalesce into one comment record only within
a category, determined by the byte after the `//` introducer: ordinary
comments (whitespace, or the end of the line or file), `//@...`
directives, and invalid introducers. A transition between categories
starts a new record, so a directive next to a comment block is its own
comment, while all the invalid spellings lump together to keep the
diagnostic noise at one per run.

The category boundary also fixes a lost directive: the invalid-comment
bulk skip compared only the `//` prefix, so `//!x` directly above
`//@dump-sem-ir-begin` absorbed the directive line and its side effect
was never recorded. Invalid comment runs now skip line by line (they
start from a diagnosed error, so they are not hot) and stop at a
whitespace or `@` introducer; the SIMD bulk skip handles only ordinary
comment blocks, whose prefix comparison already includes the whitespace
byte.

Nothing outside the lexer and the formatter reads comment records, and
lex dumps do not include comments, so no other behavior changes.

Assisted-by: Claude Code
This commit is contained in:
Chandler Carruth
2026-07-15 03:21:04 +00:00
committed by GitHub
parent 1c1870cd10
commit 474090f439
3 changed files with 236 additions and 31 deletions
+128 -8
View File
@@ -1163,6 +1163,109 @@ TEST_F(LexerTest, TrailingCommentAfterMultiLineString) {
EXPECT_TRUE(buffer.IsTrailingComment(CommentIndex(0)));
}
TEST_F(LexerTest, DirectiveComments) {
// A `//@...` directive line is consumed for its tooling side effects and is
// also recorded as a comment: the tokens and comments together reconstruct
// the source, so tooling such as the formatter preserves the directive.
auto& buffer = compile_helper_.GetTokenizedBuffer(
"//@include-in-dumps\n"
"\n"
"//@dump-sem-ir-begin\n"
"var x: i32 = 0;\n"
"//@dump-sem-ir-end\n");
EXPECT_FALSE(buffer.has_errors());
EXPECT_TRUE(buffer.has_include_in_dumps());
EXPECT_TRUE(buffer.has_dump_sem_ir_ranges());
ASSERT_THAT(buffer.comments_size(), Eq(3));
EXPECT_THAT(buffer.GetCommentText(CommentIndex(0)),
Eq("//@include-in-dumps\n"));
EXPECT_THAT(buffer.GetCommentText(CommentIndex(1)),
Eq("//@dump-sem-ir-begin\n"));
EXPECT_THAT(buffer.GetCommentText(CommentIndex(2)),
Eq("//@dump-sem-ir-end\n"));
EXPECT_FALSE(buffer.IsTrailingComment(CommentIndex(0)));
// Adjacent full-line comments coalesce only within a category: a directive
// next to an ordinary comment is its own record, while adjacent directives
// share one.
auto& adjacent = compile_helper_.GetTokenizedBuffer(
"// Dump this file.\n"
"//@include-in-dumps\n"
"//@dump-sem-ir-begin\n"
"var x: i32 = 0;\n"
"//@dump-sem-ir-end\n");
EXPECT_FALSE(adjacent.has_errors());
EXPECT_TRUE(adjacent.has_include_in_dumps());
ASSERT_THAT(adjacent.comments_size(), Eq(3));
EXPECT_THAT(adjacent.GetCommentText(CommentIndex(0)),
Eq("// Dump this file.\n"));
EXPECT_THAT(adjacent.GetCommentText(CommentIndex(1)),
Eq("//@include-in-dumps\n//@dump-sem-ir-begin\n"));
EXPECT_THAT(adjacent.GetCommentText(CommentIndex(2)),
Eq("//@dump-sem-ir-end\n"));
}
TEST_F(LexerTest, DirectiveAfterInvalidComment) {
// An invalid comment line does not absorb a following directive: the
// directive ends the invalid run, so its side effects are still recognized
// and it is recorded separately.
auto& buffer = compile_helper_.GetTokenizedBuffer(
"//!bad\n"
"//@dump-sem-ir-begin\n"
"var x: i32 = 0;\n"
"//@dump-sem-ir-end\n");
EXPECT_TRUE(buffer.has_errors());
EXPECT_TRUE(buffer.has_dump_sem_ir_ranges());
ASSERT_THAT(buffer.comments_size(), Eq(3));
EXPECT_THAT(buffer.GetCommentText(CommentIndex(0)), Eq("//!bad\n"));
EXPECT_THAT(buffer.GetCommentText(CommentIndex(1)),
Eq("//@dump-sem-ir-begin\n"));
}
TEST_F(LexerTest, InvalidCommentRunsLumpTogether) {
// A run of invalid comment lines lumps into one comment and one diagnostic,
// no matter which invalid byte follows each `//`; a valid comment ends the
// run and starts its own record.
Testing::MockDiagnosticConsumer consumer;
EXPECT_CALL(consumer,
HandleDiagnostic(IsSingleDiagnostic(
Diagnostics::Kind::NoWhitespaceAfterCommentIntroducer,
Diagnostics::Level::Error, 1, 3, _)));
auto& buffer = compile_helper_.GetTokenizedBuffer(
"//!one\n"
"//?two\n"
"// valid\n",
&consumer);
ASSERT_THAT(buffer.comments_size(), Eq(2));
EXPECT_THAT(buffer.GetCommentText(CommentIndex(0)), Eq("//!one\n//?two\n"));
EXPECT_THAT(buffer.GetCommentText(CommentIndex(1)), Eq("// valid\n"));
}
TEST_F(LexerTest, InvalidCommentRunAtEof) {
// An invalid run's line-by-line skip reads the byte after each line's `//`;
// these cases end the source at that read's boundary, with no trailing
// newline, to pin its bounds.
// The final line's introducer byte is the last byte of the source.
auto& tight = compile_helper_.GetTokenizedBuffer(
"//!a\n"
"//!");
EXPECT_TRUE(tight.has_errors());
ASSERT_THAT(tight.comments_size(), Eq(1));
EXPECT_THAT(tight.GetCommentText(CommentIndex(0)), Eq("//!a\n//!"));
// A bare `//` ending the source has no byte after the introducer at all:
// the run must stop before it rather than read past the end, and it lexes
// as its own valid, empty comment.
auto& bare = compile_helper_.GetTokenizedBuffer(
"//!a\n"
"//");
EXPECT_TRUE(bare.has_errors());
ASSERT_THAT(bare.comments_size(), Eq(2));
EXPECT_THAT(bare.GetCommentText(CommentIndex(0)), Eq("//!a\n"));
EXPECT_THAT(bare.GetCommentText(CommentIndex(1)), Eq("//"));
}
TEST_F(LexerTest, DiagnosticWhitespace) {
Testing::MockDiagnosticConsumer consumer;
EXPECT_CALL(consumer,
@@ -1318,14 +1421,14 @@ TEST_F(LexerTest, MultipleComments) {
{4}
x
)";
constexpr llvm::StringLiteral Comments[] = {
constexpr llvm::StringLiteral Groups[] = {
// NOLINTNEXTLINE(bugprone-suspicious-missing-comma)
"// This comment should be possible to parse with SIMD.\n"
"// This one too.\n",
"// This one as well, though it's a different indent.\n"
" // And mixes indent.\n"
" // And mixes indent more.\n",
"// This is one comment:\n"
"// This is several comments:\n"
"//Invalid\n"
"// Valid\n"
"//Invalid\n"
@@ -1334,18 +1437,35 @@ x
"//\n"
"// Valid\n",
"// This uses a high indent, which stops SIMD.\n", "//\n"};
std::string source = llvm::formatv(Format, Comments[0], Comments[1],
Comments[2], Comments[3], Comments[4])
std::string source = llvm::formatv(Format, Groups[0], Groups[1], Groups[2],
Groups[3], Groups[4])
.str();
auto& buffer = compile_helper_.GetTokenizedBuffer(source);
EXPECT_TRUE(buffer.has_errors());
EXPECT_THAT(buffer.comments_size(), Eq(std::size(Comments)));
for (int i :
llvm::seq(std::min<int>(buffer.comments_size(), std::size(Comments)))) {
// The third group splits at each transition between ordinary and invalid
// comment introducers; the other groups each form one comment.
constexpr llvm::StringLiteral ExpectedComments[] = {
// NOLINTNEXTLINE(bugprone-suspicious-missing-comma)
"// This comment should be possible to parse with SIMD.\n"
"// This one too.\n",
"// This one as well, though it's a different indent.\n"
" // And mixes indent.\n"
" // And mixes indent more.\n",
"// This is several comments:\n", "//Invalid\n", "// Valid\n",
"//Invalid\n",
// NOLINTNEXTLINE(bugprone-suspicious-missing-comma)
"//\n"
"// Valid\n"
"//\n"
"// Valid\n",
"// This uses a high indent, which stops SIMD.\n", "//\n"};
EXPECT_THAT(buffer.comments_size(), Eq(std::size(ExpectedComments)));
for (int i : llvm::seq(std::min<int>(buffer.comments_size(),
std::size(ExpectedComments)))) {
EXPECT_THAT(buffer.GetCommentText(CommentIndex(i)).str(),
testing::StrEq(Comments[i]));
testing::StrEq(ExpectedComments[i]));
}
EXPECT_THAT(buffer, HasTokens(llvm::ArrayRef<ExpectedToken>{
{.kind = TokenKind::FileStart},