mirror of
https://github.com/carbon-language/carbon-lang.git
synced 2026-10-02 22:02:51 +01:00
Reject invalid string literal whitespace on unescape (#793)
This is based on discussion on #732: that we should probably parse the invalid whitespace, then reject it as part of string validation, rather than having different parses. I worry the question of "how is this parsed" may lead to subtly unexpected results if we aren't consistent, so I'm switching the logic from the lexer to the unescape library (and also adjusting the list of rejected whitespace).
This commit is contained in:
+61
-51
@@ -27,58 +27,68 @@ auto UnescapeStringLiteral(llvm::StringRef source)
|
||||
size_t i = 0;
|
||||
while (i < source.size()) {
|
||||
char c = source[i];
|
||||
if (c == '\\') {
|
||||
++i;
|
||||
if (i == source.size()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
switch (source[i]) {
|
||||
case 'n':
|
||||
ret.push_back('\n');
|
||||
break;
|
||||
case 'r':
|
||||
ret.push_back('\r');
|
||||
break;
|
||||
case 't':
|
||||
ret.push_back('\t');
|
||||
break;
|
||||
case '0':
|
||||
if (i + 1 < source.size() && llvm::isDigit(source[i + 1])) {
|
||||
// \0[0-9] is reserved.
|
||||
return std::nullopt;
|
||||
}
|
||||
ret.push_back('\0');
|
||||
break;
|
||||
case '"':
|
||||
ret.push_back('"');
|
||||
break;
|
||||
case '\'':
|
||||
ret.push_back('\'');
|
||||
break;
|
||||
case '\\':
|
||||
ret.push_back('\\');
|
||||
break;
|
||||
case 'x': {
|
||||
i += 2;
|
||||
if (i >= source.size()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
std::optional<char> c1 = FromHex(source[i - 1]);
|
||||
std::optional<char> c2 = FromHex(source[i]);
|
||||
if (c1 == std::nullopt || c2 == std::nullopt) {
|
||||
return std::nullopt;
|
||||
}
|
||||
ret.push_back(16 * *c1 + *c2);
|
||||
break;
|
||||
}
|
||||
case 'u':
|
||||
FATAL() << "\\u is not yet supported in string literals";
|
||||
default:
|
||||
// Unsupported.
|
||||
switch (c) {
|
||||
case '\\':
|
||||
++i;
|
||||
if (i == source.size()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
} else {
|
||||
ret.push_back(c);
|
||||
}
|
||||
switch (source[i]) {
|
||||
case 'n':
|
||||
ret.push_back('\n');
|
||||
break;
|
||||
case 'r':
|
||||
ret.push_back('\r');
|
||||
break;
|
||||
case 't':
|
||||
ret.push_back('\t');
|
||||
break;
|
||||
case '0':
|
||||
if (i + 1 < source.size() && llvm::isDigit(source[i + 1])) {
|
||||
// \0[0-9] is reserved.
|
||||
return std::nullopt;
|
||||
}
|
||||
ret.push_back('\0');
|
||||
break;
|
||||
case '"':
|
||||
ret.push_back('"');
|
||||
break;
|
||||
case '\'':
|
||||
ret.push_back('\'');
|
||||
break;
|
||||
case '\\':
|
||||
ret.push_back('\\');
|
||||
break;
|
||||
case 'x': {
|
||||
i += 2;
|
||||
if (i >= source.size()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
std::optional<char> c1 = FromHex(source[i - 1]);
|
||||
std::optional<char> c2 = FromHex(source[i]);
|
||||
if (c1 == std::nullopt || c2 == std::nullopt) {
|
||||
return std::nullopt;
|
||||
}
|
||||
ret.push_back(16 * *c1 + *c2);
|
||||
break;
|
||||
}
|
||||
case 'u':
|
||||
FATAL() << "\\u is not yet supported in string literals";
|
||||
default:
|
||||
// Unsupported.
|
||||
return std::nullopt;
|
||||
}
|
||||
break;
|
||||
|
||||
case '\t':
|
||||
// Disallow non-` ` horizontal whitespace:
|
||||
// https://github.com/carbon-language/carbon-lang/blob/trunk/docs/design/lexical_conventions/whitespace.md
|
||||
// TODO: This doesn't handle unicode whitespace.
|
||||
return std::nullopt;
|
||||
|
||||
default:
|
||||
ret.push_back(c);
|
||||
break;
|
||||
}
|
||||
++i;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user