Fix an out-of-bounds read in TokenizedBuffer::IsRawIdentifier. (#7434)

`IsRawIdentifier` checked `token_text.starts_with("r#")` and then read
`token_text[2]`, but `starts_with` only guarantees a length of two. An
`r` identifier immediately followed by `#` at the end of the source --
so the token text is exactly `r#` -- made the `token_text[2]` read run
off the end. Guard on the length first.

Found by fuzzing. The read is reached only via `GetTokenText`, so the
parser fuzzer, which does not request token text, never hit it.

Assisted-by: Claude Code
This commit is contained in:
Chandler Carruth
2026-06-29 20:06:39 +00:00
committed by GitHub
parent 7a7aefe486
commit fd62f5a58d
2 changed files with 14 additions and 1 deletions
+13
View File
@@ -807,6 +807,19 @@ TEST_F(LexerTest, Identifiers) {
}));
}
TEST_F(LexerTest, RawIdentifierIntroducerAtEndOfFile) {
// `r#` at the very end of the source is not a raw identifier -- there is no
// identifier after the `#` -- but the leading `r` is still an identifier
// whose text begins with `r#`. Computing that text checks for the raw form
// and must not read past the end of the buffer.
auto& buffer = compile_helper_.GetTokenizedBuffer("r#");
for (TokenIndex token : buffer.tokens()) {
if (buffer.GetKind(token) == TokenKind::Identifier) {
EXPECT_EQ(buffer.GetTokenText(token), "r");
}
}
}
TEST_F(LexerTest, StringLiterals) {
llvm::StringLiteral testcase = R"(
"hello world\n"