Fix an out-of-bounds read in TokenizedBuffer::IsRawIdentifier. (#7434)

`IsRawIdentifier` checked `token_text.starts_with("r#")` and then read
`token_text[2]`, but `starts_with` only guarantees a length of two. An
`r` identifier immediately followed by `#` at the end of the source --
so the token text is exactly `r#` -- made the `token_text[2]` read run
off the end. Guard on the length first.

Found by fuzzing. The read is reached only via `GetTokenText`, so the
parser fuzzer, which does not request token text, never hit it.

Assisted-by: Claude Code
This commit is contained in:
Chandler Carruth
2026-06-29 20:06:39 +00:00
committed by GitHub
parent 7a7aefe486
commit fd62f5a58d
2 changed files with 14 additions and 1 deletions
+1 -1
View File
@@ -192,7 +192,7 @@ auto TokenizedBuffer::IsRawIdentifier(TokenIndex token) const -> bool {
// starting with `#`. It suffices to check that character is the first
// character of the identifier.
auto token_text = source_->text().substr(token_info.byte_offset());
return token_text.starts_with("r#") &&
return token_text.size() > 2 && token_text.starts_with("r#") &&
token_text[2] ==
value_stores_->identifiers().Get(token_info.ident_id()).front();
}
+13
View File
@@ -807,6 +807,19 @@ TEST_F(LexerTest, Identifiers) {
}));
}
TEST_F(LexerTest, RawIdentifierIntroducerAtEndOfFile) {
// `r#` at the very end of the source is not a raw identifier -- there is no
// identifier after the `#` -- but the leading `r` is still an identifier
// whose text begins with `r#`. Computing that text checks for the raw form
// and must not read past the end of the buffer.
auto& buffer = compile_helper_.GetTokenizedBuffer("r#");
for (TokenIndex token : buffer.tokens()) {
if (buffer.GetKind(token) == TokenKind::Identifier) {
EXPECT_EQ(buffer.GetTokenText(token), "r");
}
}
}
TEST_F(LexerTest, StringLiterals) {
llvm::StringLiteral testcase = R"(
"hello world\n"