validate only newly read bytes of binary-format strings

get_string() validated the whole result after each call, but get_bytes()
appends to it and CBOR indefinite-length strings collect all chunks in
the same result, so every chunk re-validated everything read before it.
An input of many small chunks took quadratic time (80000 one-byte chunks,
160 KB of input, took about 7 seconds). Only the newly read bytes are
validated now, which also matches RFC 8949's requirement that every
chunk is valid UTF-8 on its own.

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-23 20:42:27 +02:00
parent 2d535eb96c
commit 93c0a34171
4 changed files with 56 additions and 10 deletions
@@ -3314,6 +3314,10 @@ class binary_reader
const NumberType len, const NumberType len,
string_t& result) string_t& result)
{ {
// get_bytes() appends to result, and CBOR indefinite-length strings
// collect all their chunks in the same result; validating only the
// newly read bytes keeps the check linear in the input size
const std::size_t old_size = result.size();
if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result)))
{ {
return false; return false;
@@ -3324,7 +3328,7 @@ class binary_reader
// right here so malformed input is caught at decode time instead of // right here so malformed input is caught at decode time instead of
// only surfacing later as a type_error.316 when the value is dumped // only surfacing later as a type_error.316 when the value is dumped
// (which would defeat allow_exceptions=false / strict discarding). // (which would defeat allow_exceptions=false / strict discarding).
if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result))) if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size)))
{ {
return sax->parse_error(chars_read, get_token_string(), return sax->parse_error(chars_read, get_token_string(),
parse_error::create(113, chars_read, parse_error::create(113, chars_read,
+7 -4
View File
@@ -109,16 +109,19 @@ MessagePack/BSON specifications all require text strings to be UTF-8), so
that malformed input is caught immediately instead of only surfacing later that malformed input is caught immediately instead of only surfacing later
as a type_error.316 when the resulting value is dumped. as a type_error.316 when the resulting value is dumped.
@param[in] s the string to check @param[in] s the string to check
@return whether @a s is valid UTF-8 @param[in] first index of the first byte to check; the bytes before it are
assumed to have been validated already and to end on a
code point boundary
@return whether @a s (from index @a first on) is valid UTF-8
*/ */
template<typename StringType> template<typename StringType>
inline bool is_valid_utf8(const StringType& s) noexcept inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept
{ {
std::uint8_t state = UTF8_ACCEPT; std::uint8_t state = UTF8_ACCEPT;
std::uint32_t codepoint = 0; std::uint32_t codepoint = 0;
for (std::size_t i = 0; i < s.size(); ++i) for (std::size_t i = first; i < s.size(); ++i)
{ {
decode(state, codepoint, static_cast<std::uint8_t>(s[i])); decode(state, codepoint, static_cast<std::uint8_t>(s[i]));
if (state == UTF8_REJECT) if (state == UTF8_REJECT)
+12 -5
View File
@@ -6105,16 +6105,19 @@ MessagePack/BSON specifications all require text strings to be UTF-8), so
that malformed input is caught immediately instead of only surfacing later that malformed input is caught immediately instead of only surfacing later
as a type_error.316 when the resulting value is dumped. as a type_error.316 when the resulting value is dumped.
@param[in] s the string to check @param[in] s the string to check
@return whether @a s is valid UTF-8 @param[in] first index of the first byte to check; the bytes before it are
assumed to have been validated already and to end on a
code point boundary
@return whether @a s (from index @a first on) is valid UTF-8
*/ */
template<typename StringType> template<typename StringType>
inline bool is_valid_utf8(const StringType& s) noexcept inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept
{ {
std::uint8_t state = UTF8_ACCEPT; std::uint8_t state = UTF8_ACCEPT;
std::uint32_t codepoint = 0; std::uint32_t codepoint = 0;
for (std::size_t i = 0; i < s.size(); ++i) for (std::size_t i = first; i < s.size(); ++i)
{ {
decode(state, codepoint, static_cast<std::uint8_t>(s[i])); decode(state, codepoint, static_cast<std::uint8_t>(s[i]));
if (state == UTF8_REJECT) if (state == UTF8_REJECT)
@@ -15398,6 +15401,10 @@ class binary_reader
const NumberType len, const NumberType len,
string_t& result) string_t& result)
{ {
// get_bytes() appends to result, and CBOR indefinite-length strings
// collect all their chunks in the same result; validating only the
// newly read bytes keeps the check linear in the input size
const std::size_t old_size = result.size();
if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result))) if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result)))
{ {
return false; return false;
@@ -15408,7 +15415,7 @@ class binary_reader
// right here so malformed input is caught at decode time instead of // right here so malformed input is caught at decode time instead of
// only surfacing later as a type_error.316 when the value is dumped // only surfacing later as a type_error.316 when the value is dumped
// (which would defeat allow_exceptions=false / strict discarding). // (which would defeat allow_exceptions=false / strict discarding).
if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result))) if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size)))
{ {
return sax->parse_error(chars_read, get_token_string(), return sax->parse_error(chars_read, get_token_string(),
parse_error::create(113, chars_read, parse_error::create(113, chars_read,
+32
View File
@@ -1854,6 +1854,38 @@ TEST_CASE("CBOR")
CHECK(json::from_cbor(json::to_cbor(j)) == j); CHECK(json::from_cbor(json::to_cbor(j)) == j);
} }
SECTION("invalid UTF-8 in indefinite-length string")
{
json _;
// every chunk must be valid UTF-8 on its own (RFC 8949, Section
// 3.2.3), so a code point split across two chunks is rejected
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}), true, false).is_discarded());
// an ill-formed later chunk is rejected after valid ones
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
// valid multi-byte chunks are accepted
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6");
}
SECTION("many chunks in indefinite-length string")
{
// only the newly read chunk is validated, not the whole string
// collected so far; validating the latter made this input take
// quadratic time (about ten seconds for 100000 chunks)
constexpr std::size_t chunks = 100000;
std::vector<uint8_t> v{0x7f};
for (std::size_t i = 0; i < chunks; ++i)
{
v.push_back(0x61);
v.push_back('a');
}
v.push_back(0xff);
CHECK(json::from_cbor(v) == std::string(chunks, 'a'));
}
SECTION("strict mode") SECTION("strict mode")
{ {
std::vector<uint8_t> const vec = {0xf6, 0xf6}; std::vector<uint8_t> const vec = {0xf6, 0xf6};