Support lexing characters (#5893)

Adapts `StringLiteral` to lex characters. Adds a `CharLiteral` token,
which contains a `CharLiteralValue` which is a straight unicode code
point (suggested by zygoloid).

---------

Co-authored-by: Richard Smith <richard@metafoo.co.uk>
This commit is contained in:
Jon Ross-Perkins
2025-08-01 18:24:03 +00:00
committed by GitHub
co-authored by Richard Smith
parent c707a6deaa
commit cae8aa3adf
14 changed files with 314 additions and 79 deletions
+10 -1
View File
@@ -82,7 +82,8 @@ auto TokenizedBuffer::GetTokenText(TokenIndex token) const -> llvm::StringRef {
// Refer back to the source text to find the original spelling, including
// escape sequences etc.
if (token_info.kind() == TokenKind::StringLiteral) {
if (token_info.kind() == TokenKind::StringLiteral ||
token_info.kind() == TokenKind::CharLiteral) {
std::optional<StringLiteral> relexed_token =
StringLiteral::Lex(source_->text().substr(token_info.byte_offset()));
CARBON_CHECK(relexed_token, "Could not reform string literal token.");
@@ -137,6 +138,14 @@ auto TokenizedBuffer::GetStringLiteralValue(TokenIndex token) const
return token_info.string_literal_id();
}
auto TokenizedBuffer::GetCharLiteralValue(TokenIndex token) const
-> CharLiteralValue {
const auto& token_info = token_infos_.Get(token);
CARBON_CHECK(token_info.kind() == TokenKind::CharLiteral, "{0}",
token_info.kind());
return token_info.char_literal();
}
auto TokenizedBuffer::GetTypeLiteralSize(TokenIndex token) const -> IntId {
const auto& token_info = token_infos_.Get(token);
CARBON_CHECK(token_info.kind().is_sized_type_literal(), "{0}",